{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CQCQLA5SNHFE6RKHS7VEWMT6Q7","short_pith_number":"pith:CQCQLA5S","schema_version":"1.0","canonical_sha256":"14050583b269ca4f454797ea4b327e87d915921b92f5774bda4d0420c2836d1e","source":{"kind":"arxiv","id":"2502.02153","version":1},"attestation_state":"computed","paper":{"title":"Vulnerability Mitigation for Safety-Aligned Language Models via Debiasing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Akifumi Wachi, Rei Sato, Takumi Tanabe, Thien Q. Tran, Youhei Akimoto","submitted_at":"2025-02-04T09:31:54Z","abstract_excerpt":"Safety alignment is an essential research topic for real-world AI applications. Despite the multifaceted nature of safety and trustworthiness in AI, current safety alignment methods often focus on a comprehensive notion of safety. By carefully assessing models from the existing safety-alignment methods, we found that, while they generally improved overall safety performance, they failed to ensure safety in specific categories. Our study first identified the difficulty of eliminating such vulnerabilities without sacrificing the model's helpfulness. We observed that, while smaller KL penalty par"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.02153","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-02-04T09:31:54Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"249bcc5f8962fedd422af2388faeba46dfe374e772f45ae84a7a573586b20ed8","abstract_canon_sha256":"12ffbcc2c2c45db6f78dba515b9e0ce78ec4b138933fd5d0e682c9b856f026ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:24.827020Z","signature_b64":"aG1qekwf+GAIm5d8IO3VKqiR/I6HNZjueOR4p9jXxrkuylyv5bHOF2Cv7K97ucstYsefz3nn5W6hvesiVwwGAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"14050583b269ca4f454797ea4b327e87d915921b92f5774bda4d0420c2836d1e","last_reissued_at":"2026-07-05T10:09:24.826566Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:24.826566Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vulnerability Mitigation for Safety-Aligned Language Models via Debiasing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Akifumi Wachi, Rei Sato, Takumi Tanabe, Thien Q. Tran, Youhei Akimoto","submitted_at":"2025-02-04T09:31:54Z","abstract_excerpt":"Safety alignment is an essential research topic for real-world AI applications. Despite the multifaceted nature of safety and trustworthiness in AI, current safety alignment methods often focus on a comprehensive notion of safety. By carefully assessing models from the existing safety-alignment methods, we found that, while they generally improved overall safety performance, they failed to ensure safety in specific categories. Our study first identified the difficulty of eliminating such vulnerabilities without sacrificing the model's helpfulness. We observed that, while smaller KL penalty par"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02153","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.02153/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.02153","created_at":"2026-07-05T10:09:24.826622+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.02153v1","created_at":"2026-07-05T10:09:24.826622+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02153","created_at":"2026-07-05T10:09:24.826622+00:00"},{"alias_kind":"pith_short_12","alias_value":"CQCQLA5SNHFE","created_at":"2026-07-05T10:09:24.826622+00:00"},{"alias_kind":"pith_short_16","alias_value":"CQCQLA5SNHFE6RKH","created_at":"2026-07-05T10:09:24.826622+00:00"},{"alias_kind":"pith_short_8","alias_value":"CQCQLA5S","created_at":"2026-07-05T10:09:24.826622+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CQCQLA5SNHFE6RKHS7VEWMT6Q7","json":"https://pith.science/pith/CQCQLA5SNHFE6RKHS7VEWMT6Q7.json","graph_json":"https://pith.science/api/pith-number/CQCQLA5SNHFE6RKHS7VEWMT6Q7/graph.json","events_json":"https://pith.science/api/pith-number/CQCQLA5SNHFE6RKHS7VEWMT6Q7/events.json","paper":"https://pith.science/paper/CQCQLA5S"},"agent_actions":{"view_html":"https://pith.science/pith/CQCQLA5SNHFE6RKHS7VEWMT6Q7","download_json":"https://pith.science/pith/CQCQLA5SNHFE6RKHS7VEWMT6Q7.json","view_paper":"https://pith.science/paper/CQCQLA5S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.02153&json=true","fetch_graph":"https://pith.science/api/pith-number/CQCQLA5SNHFE6RKHS7VEWMT6Q7/graph.json","fetch_events":"https://pith.science/api/pith-number/CQCQLA5SNHFE6RKHS7VEWMT6Q7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CQCQLA5SNHFE6RKHS7VEWMT6Q7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CQCQLA5SNHFE6RKHS7VEWMT6Q7/action/storage_attestation","attest_author":"https://pith.science/pith/CQCQLA5SNHFE6RKHS7VEWMT6Q7/action/author_attestation","sign_citation":"https://pith.science/pith/CQCQLA5SNHFE6RKHS7VEWMT6Q7/action/citation_signature","submit_replication":"https://pith.science/pith/CQCQLA5SNHFE6RKHS7VEWMT6Q7/action/replication_record"}},"created_at":"2026-07-05T10:09:24.826622+00:00","updated_at":"2026-07-05T10:09:24.826622+00:00"}