{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZXET5ILLNMMTT7VKL32T3BZ6OP","short_pith_number":"pith:ZXET5ILL","schema_version":"1.0","canonical_sha256":"cdc93ea16b6b1939feaa5ef53d873e73e6f4710920eafa11d9d2cb3168fc4c99","source":{"kind":"arxiv","id":"2505.20809","version":1},"attestation_state":"computed","paper":{"title":"Improved Representation Steering for Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aryaman Arora, Christopher D. Manning, Christopher Potts, Qinan Yu, Zhengxuan Wu","submitted_at":"2025-05-27T07:16:40Z","abstract_excerpt":"Steering methods for language models (LMs) seek to provide fine-grained and interpretable control over model generations by variously changing model inputs, weights, or representations to adjust behavior. Recent work has shown that adjusting weights or representations is often less effective than steering by prompting, for instance when wanting to introduce or suppress a particular concept. We demonstrate how to improve representation steering via our new Reference-free Preference Steering (RePS), a bidirectional preference-optimization objective that jointly does concept steering and suppress"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.20809","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-27T07:16:40Z","cross_cats_sorted":[],"title_canon_sha256":"69703b7bd8b3186c967ffb4e3b0401097e3551fb926ad4075e81374e7dcdced8","abstract_canon_sha256":"cfcacb924a867d3f6274a954d9bc3c5415e75c95ba8b0d015074efc837625457"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:19.221689Z","signature_b64":"7SXitHBN9d09dq/T7/k234k4u9YRwSDInqav6ve1/hlsBsKD5QZIWruikuOOxJMH/2CTm4oQNwKb1h4rgXVmCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cdc93ea16b6b1939feaa5ef53d873e73e6f4710920eafa11d9d2cb3168fc4c99","last_reissued_at":"2026-07-05T11:10:19.221225Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:19.221225Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improved Representation Steering for Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aryaman Arora, Christopher D. Manning, Christopher Potts, Qinan Yu, Zhengxuan Wu","submitted_at":"2025-05-27T07:16:40Z","abstract_excerpt":"Steering methods for language models (LMs) seek to provide fine-grained and interpretable control over model generations by variously changing model inputs, weights, or representations to adjust behavior. Recent work has shown that adjusting weights or representations is often less effective than steering by prompting, for instance when wanting to introduce or suppress a particular concept. We demonstrate how to improve representation steering via our new Reference-free Preference Steering (RePS), a bidirectional preference-optimization objective that jointly does concept steering and suppress"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.20809","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.20809/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.20809","created_at":"2026-07-05T11:10:19.221283+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.20809v1","created_at":"2026-07-05T11:10:19.221283+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.20809","created_at":"2026-07-05T11:10:19.221283+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZXET5ILLNMMT","created_at":"2026-07-05T11:10:19.221283+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZXET5ILLNMMTT7VK","created_at":"2026-07-05T11:10:19.221283+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZXET5ILL","created_at":"2026-07-05T11:10:19.221283+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03093","citing_title":"Decomposing how prompting steers behavior","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16362","citing_title":"When Is Rank-1 Steering Cheap? Geometry, Granularity, and Budgeted Search","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16362","citing_title":"When Is Rank-1 Steering Cheap? Geometry, Granularity, and Budgeted Search","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21257","citing_title":"MoCo: A One-Stop Shop for Model Collaboration Research","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06225","citing_title":"Memory Inception: Latent-Space KV Cache Manipulation for Steering LLMs","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06225","citing_title":"Memory Inception: Latent-Space KV Cache Manipulation for Steering LLMs","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZXET5ILLNMMTT7VKL32T3BZ6OP","json":"https://pith.science/pith/ZXET5ILLNMMTT7VKL32T3BZ6OP.json","graph_json":"https://pith.science/api/pith-number/ZXET5ILLNMMTT7VKL32T3BZ6OP/graph.json","events_json":"https://pith.science/api/pith-number/ZXET5ILLNMMTT7VKL32T3BZ6OP/events.json","paper":"https://pith.science/paper/ZXET5ILL"},"agent_actions":{"view_html":"https://pith.science/pith/ZXET5ILLNMMTT7VKL32T3BZ6OP","download_json":"https://pith.science/pith/ZXET5ILLNMMTT7VKL32T3BZ6OP.json","view_paper":"https://pith.science/paper/ZXET5ILL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.20809&json=true","fetch_graph":"https://pith.science/api/pith-number/ZXET5ILLNMMTT7VKL32T3BZ6OP/graph.json","fetch_events":"https://pith.science/api/pith-number/ZXET5ILLNMMTT7VKL32T3BZ6OP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZXET5ILLNMMTT7VKL32T3BZ6OP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZXET5ILLNMMTT7VKL32T3BZ6OP/action/storage_attestation","attest_author":"https://pith.science/pith/ZXET5ILLNMMTT7VKL32T3BZ6OP/action/author_attestation","sign_citation":"https://pith.science/pith/ZXET5ILLNMMTT7VKL32T3BZ6OP/action/citation_signature","submit_replication":"https://pith.science/pith/ZXET5ILLNMMTT7VKL32T3BZ6OP/action/replication_record"}},"created_at":"2026-07-05T11:10:19.221283+00:00","updated_at":"2026-07-05T11:10:19.221283+00:00"}