{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SIXU6TAN5V7K7UTXL5YJ4JIK5K","short_pith_number":"pith:SIXU6TAN","schema_version":"1.0","canonical_sha256":"922f4f4c0ded7eafd2775f709e250aeaad14225a5fe0def7347ac914891d26a7","source":{"kind":"arxiv","id":"2503.00177","version":1},"attestation_state":"computed","paper":{"title":"Steering Large Language Model Activations in Sparse Spaces","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ali Rahimi-Kalahroudi, Mohammad Pezeshki, Pascal Vincent, Reza Bayat, Sarath Chandar","submitted_at":"2025-02-28T20:43:45Z","abstract_excerpt":"A key challenge in AI alignment is guiding large language models (LLMs) to follow desired behaviors at test time. Activation steering, which modifies internal model activations during inference, offers a potential solution. However, prior work in dense activation spaces struggles with superposition, wherein multiple features become entangled, limiting interpretability and precise control. In contrast, sparse representations provide an untapped opportunity for more interpretable behavior modulation. In this work, we introduce sparse activation steering (SAS), a method that leverages sparse auto"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.00177","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-28T20:43:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8db1deb962c1a72d279263c8ac738a30239198d1f86a713fa2c1db4f9f322ee3","abstract_canon_sha256":"2b4e067f5ace942c0bcb0a8aef33113c9528073136aa3eb39a58ed3bea01fc09"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:18.605998Z","signature_b64":"dclsbTG9BNKBN9CcyC265FYECDX58uviOtrTOMOdBzgS6EK8L5dQrvFVso1tdLhArepHOrvGYUWkgkTkmyqyDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"922f4f4c0ded7eafd2775f709e250aeaad14225a5fe0def7347ac914891d26a7","last_reissued_at":"2026-07-05T10:22:18.605451Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:18.605451Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Steering Large Language Model Activations in Sparse Spaces","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ali Rahimi-Kalahroudi, Mohammad Pezeshki, Pascal Vincent, Reza Bayat, Sarath Chandar","submitted_at":"2025-02-28T20:43:45Z","abstract_excerpt":"A key challenge in AI alignment is guiding large language models (LLMs) to follow desired behaviors at test time. Activation steering, which modifies internal model activations during inference, offers a potential solution. However, prior work in dense activation spaces struggles with superposition, wherein multiple features become entangled, limiting interpretability and precise control. In contrast, sparse representations provide an untapped opportunity for more interpretable behavior modulation. In this work, we introduce sparse activation steering (SAS), a method that leverages sparse auto"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.00177","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.00177/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.00177","created_at":"2026-07-05T10:22:18.605512+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.00177v1","created_at":"2026-07-05T10:22:18.605512+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.00177","created_at":"2026-07-05T10:22:18.605512+00:00"},{"alias_kind":"pith_short_12","alias_value":"SIXU6TAN5V7K","created_at":"2026-07-05T10:22:18.605512+00:00"},{"alias_kind":"pith_short_16","alias_value":"SIXU6TAN5V7K7UTX","created_at":"2026-07-05T10:22:18.605512+00:00"},{"alias_kind":"pith_short_8","alias_value":"SIXU6TAN","created_at":"2026-07-05T10:22:18.605512+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23670","citing_title":"Tapered Language Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03002","citing_title":"Perplexity Can Miss SAE Feature Damage Under Quantization","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08426","citing_title":"Mechanism Design Is Not Enough: Prosocial Agents for Cooperative AI","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28664","citing_title":"Activation Steering for Synthetic Data Generation: The Role of Diversity in Downstream Safety Detection","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00726","citing_title":"Latent Reward Steering: An Adaptive Inference-Time Framework that Implicitly Promotes Cognitive Behaviors in Reasoning LLMs","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06315","citing_title":"LLM Self-Recognition: Steering and Retrieving Activation Signatures","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23040","citing_title":"Steered Generation via Gradient-Based Optimization on Sparse Query Features","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22739","citing_title":"Painless Activation Steering: An Automated, Lightweight Approach for Post-Training Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01247","citing_title":"Beyond Interpretability: When, Why, and How Sparse Autoencoders Enable Label-Free Visual Steering","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11887","citing_title":"Qwen-Scope: Turning Sparse Features into Development Tools for Large Language Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08426","citing_title":"Mechanism Design Is Not Enough: Prosocial Agents for Cooperative AI","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15789","citing_title":"A Systematic Study of Training-Free Methods for Trustworthy Large Language Models","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SIXU6TAN5V7K7UTXL5YJ4JIK5K","json":"https://pith.science/pith/SIXU6TAN5V7K7UTXL5YJ4JIK5K.json","graph_json":"https://pith.science/api/pith-number/SIXU6TAN5V7K7UTXL5YJ4JIK5K/graph.json","events_json":"https://pith.science/api/pith-number/SIXU6TAN5V7K7UTXL5YJ4JIK5K/events.json","paper":"https://pith.science/paper/SIXU6TAN"},"agent_actions":{"view_html":"https://pith.science/pith/SIXU6TAN5V7K7UTXL5YJ4JIK5K","download_json":"https://pith.science/pith/SIXU6TAN5V7K7UTXL5YJ4JIK5K.json","view_paper":"https://pith.science/paper/SIXU6TAN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.00177&json=true","fetch_graph":"https://pith.science/api/pith-number/SIXU6TAN5V7K7UTXL5YJ4JIK5K/graph.json","fetch_events":"https://pith.science/api/pith-number/SIXU6TAN5V7K7UTXL5YJ4JIK5K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SIXU6TAN5V7K7UTXL5YJ4JIK5K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SIXU6TAN5V7K7UTXL5YJ4JIK5K/action/storage_attestation","attest_author":"https://pith.science/pith/SIXU6TAN5V7K7UTXL5YJ4JIK5K/action/author_attestation","sign_citation":"https://pith.science/pith/SIXU6TAN5V7K7UTXL5YJ4JIK5K/action/citation_signature","submit_replication":"https://pith.science/pith/SIXU6TAN5V7K7UTXL5YJ4JIK5K/action/replication_record"}},"created_at":"2026-07-05T10:22:18.605512+00:00","updated_at":"2026-07-05T10:22:18.605512+00:00"}