{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UTUUUBAYBZE6H3L7OVXI3FVVUU","short_pith_number":"pith:UTUUUBAY","schema_version":"1.0","canonical_sha256":"a4e94a04180e49e3ed7f756e8d96b5a519e114b82ea0e26b98a3cd24e827079b","source":{"kind":"arxiv","id":"2309.07875","version":3},"attestation_state":"computed","paper":{"title":"Safety-Tuned LLaMAs: Lessons From Improving the Safety of Large Language Models that Follow Instructions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dan Jurafsky, Federico Bianchi, Giuseppe Attanasio, James Zou, Mirac Suzgun, Paul R\\\"ottger, Tatsunori Hashimoto","submitted_at":"2023-09-14T17:23:37Z","abstract_excerpt":"Training large language models to follow instructions makes them perform better on a wide range of tasks and generally become more helpful. However, a perfectly helpful model will follow even the most malicious instructions and readily generate harmful content. In this paper, we raise concerns over the safety of models that only emphasize helpfulness, not harmlessness, in their instruction-tuning. We show that several popular instruction-tuned models are highly unsafe. Moreover, we show that adding just 3% safety examples (a few hundred demonstrations) when fine-tuning a model like LLaMA can s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.07875","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-14T17:23:37Z","cross_cats_sorted":[],"title_canon_sha256":"225f72aa6efd61771c5d0f70a82064571b12f31a83993a1ec2afc94ba18ed885","abstract_canon_sha256":"66602fe3b9cb8a0cd37bafcb30f4bf383a2ba583d2d8b267df1075a7ed4e9162"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:57:43.614463Z","signature_b64":"MKe2G5Uuh6A2M3oDsGpU5tMKaIXUB0dgoeH2TQned/V34fc6Z5B5TN/09uqqifJ+HTHdWj/3nxD5NQLKsllWAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a4e94a04180e49e3ed7f756e8d96b5a519e114b82ea0e26b98a3cd24e827079b","last_reissued_at":"2026-07-05T07:57:43.613961Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:57:43.613961Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safety-Tuned LLaMAs: Lessons From Improving the Safety of Large Language Models that Follow Instructions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dan Jurafsky, Federico Bianchi, Giuseppe Attanasio, James Zou, Mirac Suzgun, Paul R\\\"ottger, Tatsunori Hashimoto","submitted_at":"2023-09-14T17:23:37Z","abstract_excerpt":"Training large language models to follow instructions makes them perform better on a wide range of tasks and generally become more helpful. However, a perfectly helpful model will follow even the most malicious instructions and readily generate harmful content. In this paper, we raise concerns over the safety of models that only emphasize helpfulness, not harmlessness, in their instruction-tuning. We show that several popular instruction-tuned models are highly unsafe. Moreover, we show that adding just 3% safety examples (a few hundred demonstrations) when fine-tuning a model like LLaMA can s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.07875","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.07875/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.07875","created_at":"2026-07-05T07:57:43.614020+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.07875v3","created_at":"2026-07-05T07:57:43.614020+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.07875","created_at":"2026-07-05T07:57:43.614020+00:00"},{"alias_kind":"pith_short_12","alias_value":"UTUUUBAYBZE6","created_at":"2026-07-05T07:57:43.614020+00:00"},{"alias_kind":"pith_short_16","alias_value":"UTUUUBAYBZE6H3L7","created_at":"2026-07-05T07:57:43.614020+00:00"},{"alias_kind":"pith_short_8","alias_value":"UTUUUBAY","created_at":"2026-07-05T07:57:43.614020+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06196","citing_title":"Pluralis v0.1: Towards a Multicultural, Multimodal, Multilingual Benchmark for AI Risk and Reliability","ref_index":64,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03330","citing_title":"FLIPS: Instance-Fingerprinting for LLMs via Pseudo-random Sequences","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31748","citing_title":"Addressing Over-Refusal in LLMs with Competing Rewards","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22211","citing_title":"Open AI in the Wild: Adoption and Adaptation of Open Models on r/LocalLLaMA","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2408.12935","citing_title":"AI Safety Landscape for Large Language Models: Taxonomy, State-of-the-art, and Future Directions","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16737","citing_title":"Secure LLM Fine-Tuning via Safety-Aware Probing","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18795","citing_title":"HELLoRA: Hot Experts Layer-Level Low-Rank Adaptation for Mixture-of-Experts Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15239","citing_title":"Reducing the Safety Tax in LLM Safety Alignment with On-Policy Self-Distillation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2406.18495","citing_title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19790","citing_title":"Hidden Reliability Risks in Large Language Models: Systematic Identification of Precision-Induced Output Disagreements","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11882","citing_title":"On-Policy Self-Evolution via Failure Trajectories for Agentic Safety Alignment","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01034","citing_title":"A Theoretical Game of Attacks via Compositional Skills","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22089","citing_title":"Ethics Testing: Proactive Identification of Generative AI System Harms","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UTUUUBAYBZE6H3L7OVXI3FVVUU","json":"https://pith.science/pith/UTUUUBAYBZE6H3L7OVXI3FVVUU.json","graph_json":"https://pith.science/api/pith-number/UTUUUBAYBZE6H3L7OVXI3FVVUU/graph.json","events_json":"https://pith.science/api/pith-number/UTUUUBAYBZE6H3L7OVXI3FVVUU/events.json","paper":"https://pith.science/paper/UTUUUBAY"},"agent_actions":{"view_html":"https://pith.science/pith/UTUUUBAYBZE6H3L7OVXI3FVVUU","download_json":"https://pith.science/pith/UTUUUBAYBZE6H3L7OVXI3FVVUU.json","view_paper":"https://pith.science/paper/UTUUUBAY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.07875&json=true","fetch_graph":"https://pith.science/api/pith-number/UTUUUBAYBZE6H3L7OVXI3FVVUU/graph.json","fetch_events":"https://pith.science/api/pith-number/UTUUUBAYBZE6H3L7OVXI3FVVUU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UTUUUBAYBZE6H3L7OVXI3FVVUU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UTUUUBAYBZE6H3L7OVXI3FVVUU/action/storage_attestation","attest_author":"https://pith.science/pith/UTUUUBAYBZE6H3L7OVXI3FVVUU/action/author_attestation","sign_citation":"https://pith.science/pith/UTUUUBAYBZE6H3L7OVXI3FVVUU/action/citation_signature","submit_replication":"https://pith.science/pith/UTUUUBAYBZE6H3L7OVXI3FVVUU/action/replication_record"}},"created_at":"2026-07-05T07:57:43.614020+00:00","updated_at":"2026-07-05T07:57:43.614020+00:00"}