{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:XSFQBUMMCOIN4UCQCFV2RS4MQE","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"81b0ed3e00627a172a83fe7fd6b69d277668415459bb584646670d1a186728f4","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2026-06-05T07:40:34Z","title_canon_sha256":"21be594700582629c292709828dfedeb8b143d428fbbd10b4efc60f8a1ea73e1"},"schema_version":"1.0","source":{"id":"2606.07696","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2606.07696","created_at":"2026-06-09T00:04:47Z"},{"alias_kind":"arxiv_version","alias_value":"2606.07696v1","created_at":"2026-06-09T00:04:47Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.07696","created_at":"2026-06-09T00:04:47Z"},{"alias_kind":"pith_short_12","alias_value":"XSFQBUMMCOIN","created_at":"2026-06-09T00:04:47Z"},{"alias_kind":"pith_short_16","alias_value":"XSFQBUMMCOIN4UCQ","created_at":"2026-06-09T00:04:47Z"},{"alias_kind":"pith_short_8","alias_value":"XSFQBUMM","created_at":"2026-06-09T00:04:47Z"}],"graph_snapshots":[{"event_id":"sha256:3d786f202eda39429ca209f77c1581728d536343da63ebf5d647238f10da5dbd","target":"graph","created_at":"2026-06-09T00:04:47Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2606.07696/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Activation steering has become a popular training-free method to control LLM behavior by injecting precomputed direction vectors into the model's residual stream at inference time. Yet its robustness to realistic input variation remains unstudied. We present the first systematic evaluation of activation steering robustness under adversarial text perturbations on the inputs, covering four extraction methods, three attack strategies, six personas from Anthropic Model-Written Evaluation Dataset, and five models ranging from 1.5B to 30B parameters. Attacks succeed broadly across all settings: dire","authors_text":"Kien Le, Thai Le","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2026-06-05T07:40:34Z","title":"Adversarial Robustness of Activation Steering in Large Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.07696","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:123008e69e42e6e0f28eab9184ad52df54cc1bdb4558dcd6abaa3fbd99acf0b8","target":"record","created_at":"2026-06-09T00:04:47Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"81b0ed3e00627a172a83fe7fd6b69d277668415459bb584646670d1a186728f4","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2026-06-05T07:40:34Z","title_canon_sha256":"21be594700582629c292709828dfedeb8b143d428fbbd10b4efc60f8a1ea73e1"},"schema_version":"1.0","source":{"id":"2606.07696","kind":"arxiv","version":1}},"canonical_sha256":"bc8b00d18c1390de5050116ba8cb8c81203a0e7a6836ccb51b60d463285f6d5c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"bc8b00d18c1390de5050116ba8cb8c81203a0e7a6836ccb51b60d463285f6d5c","first_computed_at":"2026-06-09T00:04:47.085043Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-09T00:04:47.085043Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ePPXtt58YtdRCZQEbqCV6tqh4Noqgs3I+czbTv28vjrSL2h6NAGef6os50oI2ZS9XAlJqtnSclQI3ZHCczBVAQ==","signature_status":"signed_v1","signed_at":"2026-06-09T00:04:47.085508Z","signed_message":"canonical_sha256_bytes"},"source_id":"2606.07696","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:123008e69e42e6e0f28eab9184ad52df54cc1bdb4558dcd6abaa3fbd99acf0b8","sha256:3d786f202eda39429ca209f77c1581728d536343da63ebf5d647238f10da5dbd"],"state_sha256":"0a3490358caaba86e4e71d0ad0a0868fff031cb1b99340a5749f6732fa079fd5"}