{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZRG5XE7D7CD7DELRSHXVUQ5KES","short_pith_number":"pith:ZRG5XE7D","schema_version":"1.0","canonical_sha256":"cc4ddb93e3f887f1917191ef5a43aa248f149b2af794c4f11913d278bf2a35ff","source":{"kind":"arxiv","id":"2502.18862","version":2},"attestation_state":"computed","paper":{"title":"One-shot Optimized Steering Vectors Mediate Safety-relevant Behaviors in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Arman Cohan, Jacob Dunefsky","submitted_at":"2025-02-26T06:13:01Z","abstract_excerpt":"Steering vectors (SVs) have emerged as a promising approach for interpreting and controlling LLMs, but current methods typically require large contrastive datasets that are often impractical to construct and may capture spurious correlations. We propose directly optimizing SVs through gradient descent on a single training example, and systematically investigate how these SVs generalize. We consider several SV optimization techniques and find that the resulting SVs effectively mediate safety-relevant behaviors in multiple models. Indeed, in experiments on an alignment-faking model, we are able "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.18862","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-26T06:13:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bb54b95d6e36180d159284fc4d4c5b9ea496967eb18850dc31845838d2d4a3ad","abstract_canon_sha256":"b9422046995947982d1cb05ab23fa056e2fc80e67b49f799166c1ad8eceec44a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:52:59.758829Z","signature_b64":"qkdQmQQAE8Gy10BxDp+d3AOdt34mfLujdyY1C0bfOKuidPIXaY99LTgHG/3NDejqJ5yEA5GAjvYFeOgYJbcoAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc4ddb93e3f887f1917191ef5a43aa248f149b2af794c4f11913d278bf2a35ff","last_reissued_at":"2026-07-05T11:52:59.758368Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:52:59.758368Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"One-shot Optimized Steering Vectors Mediate Safety-relevant Behaviors in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Arman Cohan, Jacob Dunefsky","submitted_at":"2025-02-26T06:13:01Z","abstract_excerpt":"Steering vectors (SVs) have emerged as a promising approach for interpreting and controlling LLMs, but current methods typically require large contrastive datasets that are often impractical to construct and may capture spurious correlations. We propose directly optimizing SVs through gradient descent on a single training example, and systematically investigate how these SVs generalize. We consider several SV optimization techniques and find that the resulting SVs effectively mediate safety-relevant behaviors in multiple models. Indeed, in experiments on an alignment-faking model, we are able "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.18862","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.18862/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.18862","created_at":"2026-07-05T11:52:59.758426+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.18862v2","created_at":"2026-07-05T11:52:59.758426+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.18862","created_at":"2026-07-05T11:52:59.758426+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZRG5XE7D7CD7","created_at":"2026-07-05T11:52:59.758426+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZRG5XE7D7CD7DELR","created_at":"2026-07-05T11:52:59.758426+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZRG5XE7D","created_at":"2026-07-05T11:52:59.758426+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26155","citing_title":"Detecting and Controlling Sycophancy with Cascading Linear Features","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08682","citing_title":"Activation Steering Induces Emergent Misalignment: A More Comprehensive Evaluation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07696","citing_title":"Adversarial Robustness of Activation Steering in Large Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20286","citing_title":"Adaptive Probe-based Steering for Robust LLM Jailbreaking","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12798","citing_title":"Emergent and Subliminal Misalignment Through the Lens of Data-Mediated Transfer","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12199","citing_title":"Overtrained, Not Misaligned","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25783","citing_title":"Subliminal Steering: Stronger Encoding of Hidden Signals","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZRG5XE7D7CD7DELRSHXVUQ5KES","json":"https://pith.science/pith/ZRG5XE7D7CD7DELRSHXVUQ5KES.json","graph_json":"https://pith.science/api/pith-number/ZRG5XE7D7CD7DELRSHXVUQ5KES/graph.json","events_json":"https://pith.science/api/pith-number/ZRG5XE7D7CD7DELRSHXVUQ5KES/events.json","paper":"https://pith.science/paper/ZRG5XE7D"},"agent_actions":{"view_html":"https://pith.science/pith/ZRG5XE7D7CD7DELRSHXVUQ5KES","download_json":"https://pith.science/pith/ZRG5XE7D7CD7DELRSHXVUQ5KES.json","view_paper":"https://pith.science/paper/ZRG5XE7D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.18862&json=true","fetch_graph":"https://pith.science/api/pith-number/ZRG5XE7D7CD7DELRSHXVUQ5KES/graph.json","fetch_events":"https://pith.science/api/pith-number/ZRG5XE7D7CD7DELRSHXVUQ5KES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZRG5XE7D7CD7DELRSHXVUQ5KES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZRG5XE7D7CD7DELRSHXVUQ5KES/action/storage_attestation","attest_author":"https://pith.science/pith/ZRG5XE7D7CD7DELRSHXVUQ5KES/action/author_attestation","sign_citation":"https://pith.science/pith/ZRG5XE7D7CD7DELRSHXVUQ5KES/action/citation_signature","submit_replication":"https://pith.science/pith/ZRG5XE7D7CD7DELRSHXVUQ5KES/action/replication_record"}},"created_at":"2026-07-05T11:52:59.758426+00:00","updated_at":"2026-07-05T11:52:59.758426+00:00"}