{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FZR2W743VE7TUISIBSGCZS76QP","short_pith_number":"pith:FZR2W743","schema_version":"1.0","canonical_sha256":"2e63ab7f9ba93f3a22480c8c2ccbfe83e2a65e384b6f977d2b6c7efe3ced695d","source":{"kind":"arxiv","id":"2404.00495","version":1},"attestation_state":"computed","paper":{"title":"Configurable Safety Tuning of Language Models with Synthetic Preference Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Victor Gallego","submitted_at":"2024-03-30T23:28:05Z","abstract_excerpt":"State-of-the-art language model fine-tuning techniques, such as Direct Preference Optimization (DPO), restrict user control by hard-coding predefined behaviors into the model. To address this, we propose a novel method, Configurable Safety Tuning (CST), that augments DPO using synthetic preference data to facilitate flexible safety configuration of LLMs at inference time. CST overcomes the constraints of vanilla DPO by introducing a system prompt specifying safety configurations, enabling LLM deployers to disable/enable safety preferences based on their need, just changing the system prompt. O"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.00495","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-30T23:28:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"54b652a1408f6c5e70c8a49366d275459223d29d86c5f261c7dc075044f7facf","abstract_canon_sha256":"33f516d20a4cce3c4a115362a742f0dace38bfaf90a1d32d8e084b8e8a0d2c10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:34.242817Z","signature_b64":"P1AVXZujhz1UZGXYwbuV8lQHRMEH8LCxjFq8e+R+/qYSursO64bNg+3xN0Oc7fHy8HvkgtiO6uv49xY1RoYLCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e63ab7f9ba93f3a22480c8c2ccbfe83e2a65e384b6f977d2b6c7efe3ced695d","last_reissued_at":"2026-07-05T08:02:34.242424Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:34.242424Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Configurable Safety Tuning of Language Models with Synthetic Preference Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Victor Gallego","submitted_at":"2024-03-30T23:28:05Z","abstract_excerpt":"State-of-the-art language model fine-tuning techniques, such as Direct Preference Optimization (DPO), restrict user control by hard-coding predefined behaviors into the model. To address this, we propose a novel method, Configurable Safety Tuning (CST), that augments DPO using synthetic preference data to facilitate flexible safety configuration of LLMs at inference time. CST overcomes the constraints of vanilla DPO by introducing a system prompt specifying safety configurations, enabling LLM deployers to disable/enable safety preferences based on their need, just changing the system prompt. O"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.00495","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.00495/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.00495","created_at":"2026-07-05T08:02:34.242477+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.00495v1","created_at":"2026-07-05T08:02:34.242477+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.00495","created_at":"2026-07-05T08:02:34.242477+00:00"},{"alias_kind":"pith_short_12","alias_value":"FZR2W743VE7T","created_at":"2026-07-05T08:02:34.242477+00:00"},{"alias_kind":"pith_short_16","alias_value":"FZR2W743VE7TUISI","created_at":"2026-07-05T08:02:34.242477+00:00"},{"alias_kind":"pith_short_8","alias_value":"FZR2W743","created_at":"2026-07-05T08:02:34.242477+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07539","citing_title":"Prompt Governance? On Governing Technologies Governed by Natural Language","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FZR2W743VE7TUISIBSGCZS76QP","json":"https://pith.science/pith/FZR2W743VE7TUISIBSGCZS76QP.json","graph_json":"https://pith.science/api/pith-number/FZR2W743VE7TUISIBSGCZS76QP/graph.json","events_json":"https://pith.science/api/pith-number/FZR2W743VE7TUISIBSGCZS76QP/events.json","paper":"https://pith.science/paper/FZR2W743"},"agent_actions":{"view_html":"https://pith.science/pith/FZR2W743VE7TUISIBSGCZS76QP","download_json":"https://pith.science/pith/FZR2W743VE7TUISIBSGCZS76QP.json","view_paper":"https://pith.science/paper/FZR2W743","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.00495&json=true","fetch_graph":"https://pith.science/api/pith-number/FZR2W743VE7TUISIBSGCZS76QP/graph.json","fetch_events":"https://pith.science/api/pith-number/FZR2W743VE7TUISIBSGCZS76QP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FZR2W743VE7TUISIBSGCZS76QP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FZR2W743VE7TUISIBSGCZS76QP/action/storage_attestation","attest_author":"https://pith.science/pith/FZR2W743VE7TUISIBSGCZS76QP/action/author_attestation","sign_citation":"https://pith.science/pith/FZR2W743VE7TUISIBSGCZS76QP/action/citation_signature","submit_replication":"https://pith.science/pith/FZR2W743VE7TUISIBSGCZS76QP/action/replication_record"}},"created_at":"2026-07-05T08:02:34.242477+00:00","updated_at":"2026-07-05T08:02:34.242477+00:00"}