{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LQFKJS6GISR5VN6FIRXCSOEHY4","short_pith_number":"pith:LQFKJS6G","schema_version":"1.0","canonical_sha256":"5c0aa4cbc644a3dab7c5446e293887c727b12aabe00bddae0c49ab1d1f3eb3a0","source":{"kind":"arxiv","id":"2501.11183","version":1},"attestation_state":"computed","paper":{"title":"Can Safety Fine-Tuning Be More Principled? Lessons Learned from Cybersecurity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CR","authors_text":"Adam Oberman, David Williams-King, Linh Le, Yoshua Bengio","submitted_at":"2025-01-19T21:49:42Z","abstract_excerpt":"As LLMs develop increasingly advanced capabilities, there is an increased need to minimize the harm that could be caused to society by certain model outputs; hence, most LLMs have safety guardrails added, for example via fine-tuning. In this paper, we argue the position that current safety fine-tuning is very similar to a traditional cat-and-mouse game (or arms race) between attackers and defenders in cybersecurity. Model jailbreaks and attacks are patched with bandaids to target the specific attack mechanism, but many similar attack vectors might remain. When defenders are not proactively com"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.11183","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-01-19T21:49:42Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b1ffb984caa88624b532a4d8b44d774d5acee6a2ff97ff67ef9bc7e5794d99b8","abstract_canon_sha256":"83e745680026e043b94daee070ad5be32a4f3ca17ee3de8c861b85461a60cd1f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:03:04.735282Z","signature_b64":"pQFXseQu05gMwYXrE4Etw0BQhv+b72AMpKRBmKMpNd4RXPi2nudo9aI7lJQD8rw2KqAGu1h8m5isqxMNNQq4CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5c0aa4cbc644a3dab7c5446e293887c727b12aabe00bddae0c49ab1d1f3eb3a0","last_reissued_at":"2026-07-05T10:03:04.734903Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:03:04.734903Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Safety Fine-Tuning Be More Principled? Lessons Learned from Cybersecurity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CR","authors_text":"Adam Oberman, David Williams-King, Linh Le, Yoshua Bengio","submitted_at":"2025-01-19T21:49:42Z","abstract_excerpt":"As LLMs develop increasingly advanced capabilities, there is an increased need to minimize the harm that could be caused to society by certain model outputs; hence, most LLMs have safety guardrails added, for example via fine-tuning. In this paper, we argue the position that current safety fine-tuning is very similar to a traditional cat-and-mouse game (or arms race) between attackers and defenders in cybersecurity. Model jailbreaks and attacks are patched with bandaids to target the specific attack mechanism, but many similar attack vectors might remain. When defenders are not proactively com"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.11183","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.11183/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.11183","created_at":"2026-07-05T10:03:04.734969+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.11183v1","created_at":"2026-07-05T10:03:04.734969+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.11183","created_at":"2026-07-05T10:03:04.734969+00:00"},{"alias_kind":"pith_short_12","alias_value":"LQFKJS6GISR5","created_at":"2026-07-05T10:03:04.734969+00:00"},{"alias_kind":"pith_short_16","alias_value":"LQFKJS6GISR5VN6F","created_at":"2026-07-05T10:03:04.734969+00:00"},{"alias_kind":"pith_short_8","alias_value":"LQFKJS6G","created_at":"2026-07-05T10:03:04.734969+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LQFKJS6GISR5VN6FIRXCSOEHY4","json":"https://pith.science/pith/LQFKJS6GISR5VN6FIRXCSOEHY4.json","graph_json":"https://pith.science/api/pith-number/LQFKJS6GISR5VN6FIRXCSOEHY4/graph.json","events_json":"https://pith.science/api/pith-number/LQFKJS6GISR5VN6FIRXCSOEHY4/events.json","paper":"https://pith.science/paper/LQFKJS6G"},"agent_actions":{"view_html":"https://pith.science/pith/LQFKJS6GISR5VN6FIRXCSOEHY4","download_json":"https://pith.science/pith/LQFKJS6GISR5VN6FIRXCSOEHY4.json","view_paper":"https://pith.science/paper/LQFKJS6G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.11183&json=true","fetch_graph":"https://pith.science/api/pith-number/LQFKJS6GISR5VN6FIRXCSOEHY4/graph.json","fetch_events":"https://pith.science/api/pith-number/LQFKJS6GISR5VN6FIRXCSOEHY4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LQFKJS6GISR5VN6FIRXCSOEHY4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LQFKJS6GISR5VN6FIRXCSOEHY4/action/storage_attestation","attest_author":"https://pith.science/pith/LQFKJS6GISR5VN6FIRXCSOEHY4/action/author_attestation","sign_citation":"https://pith.science/pith/LQFKJS6GISR5VN6FIRXCSOEHY4/action/citation_signature","submit_replication":"https://pith.science/pith/LQFKJS6GISR5VN6FIRXCSOEHY4/action/replication_record"}},"created_at":"2026-07-05T10:03:04.734969+00:00","updated_at":"2026-07-05T10:03:04.734969+00:00"}