{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NB2SUCZDJSQYLRTN5ZRSIL5SBB","short_pith_number":"pith:NB2SUCZD","schema_version":"1.0","canonical_sha256":"68752a0b234ca185c66dee63242fb2084aa3a07372843312f8bb3903a6e50daf","source":{"kind":"arxiv","id":"2405.17374","version":3},"attestation_state":"computed","paper":{"title":"Navigating the Safety Landscape: Measuring Risks in Finetuning Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Duen Horng Chau, Matthew Hull, Pin-Yu Chen, ShengYun Peng","submitted_at":"2024-05-27T17:31:56Z","abstract_excerpt":"Safety alignment is crucial to ensure that large language models (LLMs) behave in ways that align with human preferences and prevent harmful actions during inference. However, recent studies show that the alignment can be easily compromised through finetuning with only a few adversarially designed training examples. We aim to measure the risks in finetuning LLMs through navigating the LLM safety landscape. We discover a new phenomenon observed universally in the model parameter space of popular open-source LLMs, termed as \"safety basin\": random perturbations to model weights maintain the safet"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.17374","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-27T17:31:56Z","cross_cats_sorted":[],"title_canon_sha256":"c4e8ff4bec0151e1a851327c4e177b12ee1a21ee17a9740bb5a11a3130e1b564","abstract_canon_sha256":"c36947b46c11a9578f82fc1bcf4db8ba304aebc41cf8d06f4d92de81d34be231"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:28:50.342058Z","signature_b64":"u5+FFCEKE+R/sqgZ2O1bX+oJUsACReIQ4hduz4wBnPJ+XO4DXnLS1TZUTD71iU3ZyABwIX6a7hjrPj4TbNlaCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"68752a0b234ca185c66dee63242fb2084aa3a07372843312f8bb3903a6e50daf","last_reissued_at":"2026-07-05T09:28:50.341599Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:28:50.341599Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Navigating the Safety Landscape: Measuring Risks in Finetuning Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Duen Horng Chau, Matthew Hull, Pin-Yu Chen, ShengYun Peng","submitted_at":"2024-05-27T17:31:56Z","abstract_excerpt":"Safety alignment is crucial to ensure that large language models (LLMs) behave in ways that align with human preferences and prevent harmful actions during inference. However, recent studies show that the alignment can be easily compromised through finetuning with only a few adversarially designed training examples. We aim to measure the risks in finetuning LLMs through navigating the LLM safety landscape. We discover a new phenomenon observed universally in the model parameter space of popular open-source LLMs, termed as \"safety basin\": random perturbations to model weights maintain the safet"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.17374","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.17374/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.17374","created_at":"2026-07-05T09:28:50.341657+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.17374v3","created_at":"2026-07-05T09:28:50.341657+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.17374","created_at":"2026-07-05T09:28:50.341657+00:00"},{"alias_kind":"pith_short_12","alias_value":"NB2SUCZDJSQY","created_at":"2026-07-05T09:28:50.341657+00:00"},{"alias_kind":"pith_short_16","alias_value":"NB2SUCZDJSQYLRTN","created_at":"2026-07-05T09:28:50.341657+00:00"},{"alias_kind":"pith_short_8","alias_value":"NB2SUCZD","created_at":"2026-07-05T09:28:50.341657+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28525","citing_title":"A Gravitational Interpretation of Fine-Tuning Reversion","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01318","citing_title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19638","citing_title":"SafetyALFRED: Evaluating Safety-Conscious Planning of Multimodal Large Language Models","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NB2SUCZDJSQYLRTN5ZRSIL5SBB","json":"https://pith.science/pith/NB2SUCZDJSQYLRTN5ZRSIL5SBB.json","graph_json":"https://pith.science/api/pith-number/NB2SUCZDJSQYLRTN5ZRSIL5SBB/graph.json","events_json":"https://pith.science/api/pith-number/NB2SUCZDJSQYLRTN5ZRSIL5SBB/events.json","paper":"https://pith.science/paper/NB2SUCZD"},"agent_actions":{"view_html":"https://pith.science/pith/NB2SUCZDJSQYLRTN5ZRSIL5SBB","download_json":"https://pith.science/pith/NB2SUCZDJSQYLRTN5ZRSIL5SBB.json","view_paper":"https://pith.science/paper/NB2SUCZD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.17374&json=true","fetch_graph":"https://pith.science/api/pith-number/NB2SUCZDJSQYLRTN5ZRSIL5SBB/graph.json","fetch_events":"https://pith.science/api/pith-number/NB2SUCZDJSQYLRTN5ZRSIL5SBB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NB2SUCZDJSQYLRTN5ZRSIL5SBB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NB2SUCZDJSQYLRTN5ZRSIL5SBB/action/storage_attestation","attest_author":"https://pith.science/pith/NB2SUCZDJSQYLRTN5ZRSIL5SBB/action/author_attestation","sign_citation":"https://pith.science/pith/NB2SUCZDJSQYLRTN5ZRSIL5SBB/action/citation_signature","submit_replication":"https://pith.science/pith/NB2SUCZDJSQYLRTN5ZRSIL5SBB/action/replication_record"}},"created_at":"2026-07-05T09:28:50.341657+00:00","updated_at":"2026-07-05T09:28:50.341657+00:00"}