{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JCNUN3TUUHRW2S3SYVJPP27P2X","short_pith_number":"pith:JCNUN3TU","schema_version":"1.0","canonical_sha256":"489b46ee74a1e36d4b72c552f7ebefd5e135485c0737e46ec9c158a4dae4ade2","source":{"kind":"arxiv","id":"2406.05954","version":3},"attestation_state":"computed","paper":{"title":"Aligning Large Language Models with Representation Editing: A Control Perspective","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.AI","authors_text":"Chao Zhang, Haorui Wang, Kai Wang, Lingkai Kong, Rongzhi Zhang, Wenhao Mu, Yifei Zhou, Yuanqi Du, Yuchen Zhuang, Yue Song","submitted_at":"2024-06-10T01:21:31Z","abstract_excerpt":"Aligning large language models (LLMs) with human objectives is crucial for real-world applications. However, fine-tuning LLMs for alignment often suffers from unstable training and requires substantial computing resources. Test-time alignment techniques, such as prompting and guided decoding, do not modify the underlying model, and their performance remains dependent on the original model's capabilities. To address these challenges, we propose aligning LLMs through representation editing. The core of our method is to view a pre-trained autoregressive LLM as a discrete-time stochastic dynamical"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05954","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-10T01:21:31Z","cross_cats_sorted":["cs.LG","cs.SY","eess.SY"],"title_canon_sha256":"c778c8d4f631bb4549665596804def551f02cecbb0179c8841af2eca85f5a6d8","abstract_canon_sha256":"a8a670d18be158392414fd44efc3f72a04234ab01c404a40c87b74ac1f1012de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:29:49.417480Z","signature_b64":"iGDlEGyuDwWluglGCQnnR1Fq33yB6WmJZ3NouQpebdrdmA1WNH8neyUWxu+F+SvCz/jZq7O/j/fS7+vIwPOUAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"489b46ee74a1e36d4b72c552f7ebefd5e135485c0737e46ec9c158a4dae4ade2","last_reissued_at":"2026-07-05T09:29:49.416984Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:29:49.416984Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Aligning Large Language Models with Representation Editing: A Control Perspective","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.AI","authors_text":"Chao Zhang, Haorui Wang, Kai Wang, Lingkai Kong, Rongzhi Zhang, Wenhao Mu, Yifei Zhou, Yuanqi Du, Yuchen Zhuang, Yue Song","submitted_at":"2024-06-10T01:21:31Z","abstract_excerpt":"Aligning large language models (LLMs) with human objectives is crucial for real-world applications. However, fine-tuning LLMs for alignment often suffers from unstable training and requires substantial computing resources. Test-time alignment techniques, such as prompting and guided decoding, do not modify the underlying model, and their performance remains dependent on the original model's capabilities. To address these challenges, we propose aligning LLMs through representation editing. The core of our method is to view a pre-trained autoregressive LLM as a discrete-time stochastic dynamical"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05954","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05954/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05954","created_at":"2026-07-05T09:29:49.417044+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05954v3","created_at":"2026-07-05T09:29:49.417044+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05954","created_at":"2026-07-05T09:29:49.417044+00:00"},{"alias_kind":"pith_short_12","alias_value":"JCNUN3TUUHRW","created_at":"2026-07-05T09:29:49.417044+00:00"},{"alias_kind":"pith_short_16","alias_value":"JCNUN3TUUHRW2S3S","created_at":"2026-07-05T09:29:49.417044+00:00"},{"alias_kind":"pith_short_8","alias_value":"JCNUN3TU","created_at":"2026-07-05T09:29:49.417044+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04775","citing_title":"Activation Steering of Video Generation Models via Reduced-Order Linear Optimal Control","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04309","citing_title":"Activation Steering with a Feedback Controller","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19018","citing_title":"Local Linearity of LLMs Enables Activation Steering via Model-Based Linear Optimal Control","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JCNUN3TUUHRW2S3SYVJPP27P2X","json":"https://pith.science/pith/JCNUN3TUUHRW2S3SYVJPP27P2X.json","graph_json":"https://pith.science/api/pith-number/JCNUN3TUUHRW2S3SYVJPP27P2X/graph.json","events_json":"https://pith.science/api/pith-number/JCNUN3TUUHRW2S3SYVJPP27P2X/events.json","paper":"https://pith.science/paper/JCNUN3TU"},"agent_actions":{"view_html":"https://pith.science/pith/JCNUN3TUUHRW2S3SYVJPP27P2X","download_json":"https://pith.science/pith/JCNUN3TUUHRW2S3SYVJPP27P2X.json","view_paper":"https://pith.science/paper/JCNUN3TU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05954&json=true","fetch_graph":"https://pith.science/api/pith-number/JCNUN3TUUHRW2S3SYVJPP27P2X/graph.json","fetch_events":"https://pith.science/api/pith-number/JCNUN3TUUHRW2S3SYVJPP27P2X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JCNUN3TUUHRW2S3SYVJPP27P2X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JCNUN3TUUHRW2S3SYVJPP27P2X/action/storage_attestation","attest_author":"https://pith.science/pith/JCNUN3TUUHRW2S3SYVJPP27P2X/action/author_attestation","sign_citation":"https://pith.science/pith/JCNUN3TUUHRW2S3SYVJPP27P2X/action/citation_signature","submit_replication":"https://pith.science/pith/JCNUN3TUUHRW2S3SYVJPP27P2X/action/replication_record"}},"created_at":"2026-07-05T09:29:49.417044+00:00","updated_at":"2026-07-05T09:29:49.417044+00:00"}