{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TCN3ZQMSKFCIACX223VX26VVEP","short_pith_number":"pith:TCN3ZQMS","schema_version":"1.0","canonical_sha256":"989bbcc1925144800afad6eb7d7ab523d3d98b7245ebf0eb22e35375d1c29471","source":{"kind":"arxiv","id":"2405.19358","version":2},"attestation_state":"computed","paper":{"title":"Robustifying Safety-Aligned Large Language Models through Clean Data Curation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Jiacheng Liang, Muchao Ye, Xiaoqun Liu, Zhaohan Xi","submitted_at":"2024-05-24T04:50:38Z","abstract_excerpt":"Large language models (LLMs) are vulnerable when trained on datasets containing harmful content, which leads to potential jailbreaking attacks in two scenarios: the integration of harmful texts within crowdsourced data used for pre-training and direct tampering with LLMs through fine-tuning. In both scenarios, adversaries can compromise the safety alignment of LLMs, exacerbating malfunctions. Motivated by the need to mitigate these adversarial influences, our research aims to enhance safety alignment by either neutralizing the impact of malicious texts in pre-training datasets or increasing th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.19358","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-05-24T04:50:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7f4c905e141284e373c7cc18240aac1802ecb251415488e20e18d3a31f885814","abstract_canon_sha256":"a7f1ac8505e91cea9c175bff7ba5d1340da90bc7b3edbf59bbaeaf707a03b258"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:35.198459Z","signature_b64":"pxyouQx1HOr29rKMohEXrbKdOke4KCFTftK7tCTOMKmLsTljWFR5GnM/J3uJi2Dpw78MqIZWD4fXxrvPOQQgBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"989bbcc1925144800afad6eb7d7ab523d3d98b7245ebf0eb22e35375d1c29471","last_reissued_at":"2026-07-05T08:25:35.198033Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:35.198033Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robustifying Safety-Aligned Large Language Models through Clean Data Curation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Jiacheng Liang, Muchao Ye, Xiaoqun Liu, Zhaohan Xi","submitted_at":"2024-05-24T04:50:38Z","abstract_excerpt":"Large language models (LLMs) are vulnerable when trained on datasets containing harmful content, which leads to potential jailbreaking attacks in two scenarios: the integration of harmful texts within crowdsourced data used for pre-training and direct tampering with LLMs through fine-tuning. In both scenarios, adversaries can compromise the safety alignment of LLMs, exacerbating malfunctions. Motivated by the need to mitigate these adversarial influences, our research aims to enhance safety alignment by either neutralizing the impact of malicious texts in pre-training datasets or increasing th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19358","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.19358/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.19358","created_at":"2026-07-05T08:25:35.198089+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.19358v2","created_at":"2026-07-05T08:25:35.198089+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19358","created_at":"2026-07-05T08:25:35.198089+00:00"},{"alias_kind":"pith_short_12","alias_value":"TCN3ZQMSKFCI","created_at":"2026-07-05T08:25:35.198089+00:00"},{"alias_kind":"pith_short_16","alias_value":"TCN3ZQMSKFCIACX2","created_at":"2026-07-05T08:25:35.198089+00:00"},{"alias_kind":"pith_short_8","alias_value":"TCN3ZQMS","created_at":"2026-07-05T08:25:35.198089+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19890","citing_title":"Open Weight AI Models Require Proportional Evaluation Approaches","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14605","citing_title":"One Step to the Side: Why Defenses Against Malicious Finetuning Fail Under Adaptive Adversaries","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29396","citing_title":"Aligned but Fragile: Enhancing LLM Safety Robustness via Zeroth-Order Optimization","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":101,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TCN3ZQMSKFCIACX223VX26VVEP","json":"https://pith.science/pith/TCN3ZQMSKFCIACX223VX26VVEP.json","graph_json":"https://pith.science/api/pith-number/TCN3ZQMSKFCIACX223VX26VVEP/graph.json","events_json":"https://pith.science/api/pith-number/TCN3ZQMSKFCIACX223VX26VVEP/events.json","paper":"https://pith.science/paper/TCN3ZQMS"},"agent_actions":{"view_html":"https://pith.science/pith/TCN3ZQMSKFCIACX223VX26VVEP","download_json":"https://pith.science/pith/TCN3ZQMSKFCIACX223VX26VVEP.json","view_paper":"https://pith.science/paper/TCN3ZQMS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.19358&json=true","fetch_graph":"https://pith.science/api/pith-number/TCN3ZQMSKFCIACX223VX26VVEP/graph.json","fetch_events":"https://pith.science/api/pith-number/TCN3ZQMSKFCIACX223VX26VVEP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TCN3ZQMSKFCIACX223VX26VVEP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TCN3ZQMSKFCIACX223VX26VVEP/action/storage_attestation","attest_author":"https://pith.science/pith/TCN3ZQMSKFCIACX223VX26VVEP/action/author_attestation","sign_citation":"https://pith.science/pith/TCN3ZQMSKFCIACX223VX26VVEP/action/citation_signature","submit_replication":"https://pith.science/pith/TCN3ZQMSKFCIACX223VX26VVEP/action/replication_record"}},"created_at":"2026-07-05T08:25:35.198089+00:00","updated_at":"2026-07-05T08:25:35.198089+00:00"}