{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IXXD4D3VL5QYBZAPHJDSARJ47O","short_pith_number":"pith:IXXD4D3V","schema_version":"1.0","canonical_sha256":"45ee3e0f755f6180e40f3a4720453cfb883644640a19f8b03f074e981414c4fb","source":{"kind":"arxiv","id":"2411.02193","version":2},"attestation_state":"computed","paper":{"title":"Improving Steering Vectors by Targeting Sparse Autoencoder Features","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Arthur Conmy, Matthew Siu, Sviatoslav Chalnev","submitted_at":"2024-11-04T15:46:20Z","abstract_excerpt":"To control the behavior of language models, steering methods attempt to ensure that outputs of the model satisfy specific pre-defined properties. Adding steering vectors to the model is a promising method of model control that is easier than finetuning, and may be more robust than prompting. However, it can be difficult to anticipate the effects of steering vectors produced by methods such as CAA [Panickssery et al., 2024] or the direct use of SAE latents [Templeton et al., 2024]. In our work, we address this issue by using SAEs to measure the effects of steering vectors, giving us a method th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.02193","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-04T15:46:20Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"0c9447467dc1a723e83580e0370f954cd71cbbc1f8f9f4fe4dec0c42d5d8757a","abstract_canon_sha256":"24486a5506058d149ff28046d4847b0f833373cda1c6a177d9cae6486e2ef69e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:31.870916Z","signature_b64":"o8FiK560p1OaU9HqWyIphb1Y80n+T9S3xazkf72uBnQYfE0/TbWQ73jNXjr+p8xVmOJ5n2mTWCnULnNGvNPtDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45ee3e0f755f6180e40f3a4720453cfb883644640a19f8b03f074e981414c4fb","last_reissued_at":"2026-07-05T09:38:31.870460Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:31.870460Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Steering Vectors by Targeting Sparse Autoencoder Features","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Arthur Conmy, Matthew Siu, Sviatoslav Chalnev","submitted_at":"2024-11-04T15:46:20Z","abstract_excerpt":"To control the behavior of language models, steering methods attempt to ensure that outputs of the model satisfy specific pre-defined properties. Adding steering vectors to the model is a promising method of model control that is easier than finetuning, and may be more robust than prompting. However, it can be difficult to anticipate the effects of steering vectors produced by methods such as CAA [Panickssery et al., 2024] or the direct use of SAE latents [Templeton et al., 2024]. In our work, we address this issue by using SAEs to measure the effects of steering vectors, giving us a method th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.02193","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.02193/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.02193","created_at":"2026-07-05T09:38:31.870515+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.02193v2","created_at":"2026-07-05T09:38:31.870515+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.02193","created_at":"2026-07-05T09:38:31.870515+00:00"},{"alias_kind":"pith_short_12","alias_value":"IXXD4D3VL5QY","created_at":"2026-07-05T09:38:31.870515+00:00"},{"alias_kind":"pith_short_16","alias_value":"IXXD4D3VL5QYBZAP","created_at":"2026-07-05T09:38:31.870515+00:00"},{"alias_kind":"pith_short_8","alias_value":"IXXD4D3V","created_at":"2026-07-05T09:38:31.870515+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08496","citing_title":"SAEExplainer: Interpreting SAE Features with Activation-Guided Preference Optimization","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03002","citing_title":"Perplexity Can Miss SAE Feature Damage Under Quantization","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01844","citing_title":"The Cylindrical Representation Hypothesis for Language Model Steering","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28664","citing_title":"Activation Steering for Synthetic Data Generation: The Role of Diversity in Downstream Safety Detection","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28669","citing_title":"Sense Representations Are Inducible Interfaces","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23040","citing_title":"Steered Generation via Gradient-Based Optimization on Sparse Query Features","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23036","citing_title":"Multilingual Steering by Design: Multilingual Sparse Autoencoders and Principled Layer Selection","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01770","citing_title":"ReGA: Model-Based Safeguard for LLMs via Representation-Guided Abstraction","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02280","citing_title":"RACC: Representation-Aware Coverage Criteria for LLM Safety Testing","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12671","citing_title":"All Circuits Lead to Rome: Rethinking Functional Anisotropy in Circuit and Sheaf Discovery for LLMs","ref_index":139,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IXXD4D3VL5QYBZAPHJDSARJ47O","json":"https://pith.science/pith/IXXD4D3VL5QYBZAPHJDSARJ47O.json","graph_json":"https://pith.science/api/pith-number/IXXD4D3VL5QYBZAPHJDSARJ47O/graph.json","events_json":"https://pith.science/api/pith-number/IXXD4D3VL5QYBZAPHJDSARJ47O/events.json","paper":"https://pith.science/paper/IXXD4D3V"},"agent_actions":{"view_html":"https://pith.science/pith/IXXD4D3VL5QYBZAPHJDSARJ47O","download_json":"https://pith.science/pith/IXXD4D3VL5QYBZAPHJDSARJ47O.json","view_paper":"https://pith.science/paper/IXXD4D3V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.02193&json=true","fetch_graph":"https://pith.science/api/pith-number/IXXD4D3VL5QYBZAPHJDSARJ47O/graph.json","fetch_events":"https://pith.science/api/pith-number/IXXD4D3VL5QYBZAPHJDSARJ47O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IXXD4D3VL5QYBZAPHJDSARJ47O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IXXD4D3VL5QYBZAPHJDSARJ47O/action/storage_attestation","attest_author":"https://pith.science/pith/IXXD4D3VL5QYBZAPHJDSARJ47O/action/author_attestation","sign_citation":"https://pith.science/pith/IXXD4D3VL5QYBZAPHJDSARJ47O/action/citation_signature","submit_replication":"https://pith.science/pith/IXXD4D3VL5QYBZAPHJDSARJ47O/action/replication_record"}},"created_at":"2026-07-05T09:38:31.870515+00:00","updated_at":"2026-07-05T09:38:31.870515+00:00"}