{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:T5MPV7N4VPOFTQFBW32ZVQXQXK","short_pith_number":"pith:T5MPV7N4","schema_version":"1.0","canonical_sha256":"9f58fafdbcabdc59c0a1b6f59ac2f0babe38c853f4fe2cf3f1ce5bb4b4e14945","source":{"kind":"arxiv","id":"2504.18872","version":1},"attestation_state":"computed","paper":{"title":"Latent Adversarial Training Improves the Representation of Refusal","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexandra Abbas, Helios Ael Lyons, Natalia Perez-Campanero, Nora Petrova","submitted_at":"2025-04-26T09:40:31Z","abstract_excerpt":"Recent work has shown that language models' refusal behavior is primarily encoded in a single direction in their latent space, making it vulnerable to targeted attacks. Although Latent Adversarial Training (LAT) attempts to improve robustness by introducing noise during training, a key question remains: How does this noise-based training affect the underlying representation of refusal behavior? Understanding this encoding is crucial for evaluating LAT's effectiveness and limitations, just as the discovery of linear refusal directions revealed vulnerabilities in traditional supervised safety fi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.18872","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-26T09:40:31Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e647060e989f619db528ac04318a908b9aec2cdbb94893f8e041d7dfe5a7a9ce","abstract_canon_sha256":"c03db05ff7fa594385e3f489bb0b58650beb3fdb1897f2b261dc2a9366be74bf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:22.930490Z","signature_b64":"wBdBBjuUxlHFodISlCrZI+5W8Fcv9A1VLmtdu0WHwHEqtlvOFk9JqRtP/Gvdhhxykx3ZzjoieJe6frQ/Get+DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f58fafdbcabdc59c0a1b6f59ac2f0babe38c853f4fe2cf3f1ce5bb4b4e14945","last_reissued_at":"2026-07-05T10:54:22.930003Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:22.930003Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Latent Adversarial Training Improves the Representation of Refusal","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexandra Abbas, Helios Ael Lyons, Natalia Perez-Campanero, Nora Petrova","submitted_at":"2025-04-26T09:40:31Z","abstract_excerpt":"Recent work has shown that language models' refusal behavior is primarily encoded in a single direction in their latent space, making it vulnerable to targeted attacks. Although Latent Adversarial Training (LAT) attempts to improve robustness by introducing noise during training, a key question remains: How does this noise-based training affect the underlying representation of refusal behavior? Understanding this encoding is crucial for evaluating LAT's effectiveness and limitations, just as the discovery of linear refusal directions revealed vulnerabilities in traditional supervised safety fi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.18872","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.18872/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.18872","created_at":"2026-07-05T10:54:22.930063+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.18872v1","created_at":"2026-07-05T10:54:22.930063+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.18872","created_at":"2026-07-05T10:54:22.930063+00:00"},{"alias_kind":"pith_short_12","alias_value":"T5MPV7N4VPOF","created_at":"2026-07-05T10:54:22.930063+00:00"},{"alias_kind":"pith_short_16","alias_value":"T5MPV7N4VPOFTQFB","created_at":"2026-07-05T10:54:22.930063+00:00"},{"alias_kind":"pith_short_8","alias_value":"T5MPV7N4","created_at":"2026-07-05T10:54:22.930063+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T5MPV7N4VPOFTQFBW32ZVQXQXK","json":"https://pith.science/pith/T5MPV7N4VPOFTQFBW32ZVQXQXK.json","graph_json":"https://pith.science/api/pith-number/T5MPV7N4VPOFTQFBW32ZVQXQXK/graph.json","events_json":"https://pith.science/api/pith-number/T5MPV7N4VPOFTQFBW32ZVQXQXK/events.json","paper":"https://pith.science/paper/T5MPV7N4"},"agent_actions":{"view_html":"https://pith.science/pith/T5MPV7N4VPOFTQFBW32ZVQXQXK","download_json":"https://pith.science/pith/T5MPV7N4VPOFTQFBW32ZVQXQXK.json","view_paper":"https://pith.science/paper/T5MPV7N4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.18872&json=true","fetch_graph":"https://pith.science/api/pith-number/T5MPV7N4VPOFTQFBW32ZVQXQXK/graph.json","fetch_events":"https://pith.science/api/pith-number/T5MPV7N4VPOFTQFBW32ZVQXQXK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T5MPV7N4VPOFTQFBW32ZVQXQXK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T5MPV7N4VPOFTQFBW32ZVQXQXK/action/storage_attestation","attest_author":"https://pith.science/pith/T5MPV7N4VPOFTQFBW32ZVQXQXK/action/author_attestation","sign_citation":"https://pith.science/pith/T5MPV7N4VPOFTQFBW32ZVQXQXK/action/citation_signature","submit_replication":"https://pith.science/pith/T5MPV7N4VPOFTQFBW32ZVQXQXK/action/replication_record"}},"created_at":"2026-07-05T10:54:22.930063+00:00","updated_at":"2026-07-05T10:54:22.930063+00:00"}