{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ARDP5LAFFKUSZGXBNPCYJ5DMTT","short_pith_number":"pith:ARDP5LAF","schema_version":"1.0","canonical_sha256":"0446feac052aa92c9ae16bc584f46c9cec4b8dba268d51e3d8e03f6f96f9fa35","source":{"kind":"arxiv","id":"2506.04679","version":1},"attestation_state":"computed","paper":{"title":"Normative Conflicts and Shallow AI Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Rapha\\\"el Milli\\`ere","submitted_at":"2025-06-05T06:57:28Z","abstract_excerpt":"The progress of AI systems such as large language models (LLMs) raises increasingly pressing concerns about their safe deployment. This paper examines the value alignment problem for LLMs, arguing that current alignment strategies are fundamentally inadequate to prevent misuse. Despite ongoing efforts to instill norms such as helpfulness, honesty, and harmlessness in LLMs through fine-tuning based on human preferences, they remain vulnerable to adversarial attacks that exploit conflicts between these norms. I argue that this vulnerability reflects a fundamental limitation of existing alignment"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.04679","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-05T06:57:28Z","cross_cats_sorted":[],"title_canon_sha256":"1fac0a139d90a1ad1cb5ef9410c5ef21562fd91585a4c137609e8a262e9e3576","abstract_canon_sha256":"5c84da52fdd8c13cbce985f22a036b3f35486600feba4eb974d1a6197c06f08c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:18.798417Z","signature_b64":"rTi9/3NSh/mVYtc6RnQfnRs4HLtt82A+vfxER/bZzaFCJkybuG9Tnzby3oKpkU+oQalvGwfBLWgtPe0yS4iEAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0446feac052aa92c9ae16bc584f46c9cec4b8dba268d51e3d8e03f6f96f9fa35","last_reissued_at":"2026-07-05T11:16:18.798006Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:18.798006Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Normative Conflicts and Shallow AI Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Rapha\\\"el Milli\\`ere","submitted_at":"2025-06-05T06:57:28Z","abstract_excerpt":"The progress of AI systems such as large language models (LLMs) raises increasingly pressing concerns about their safe deployment. This paper examines the value alignment problem for LLMs, arguing that current alignment strategies are fundamentally inadequate to prevent misuse. Despite ongoing efforts to instill norms such as helpfulness, honesty, and harmlessness in LLMs through fine-tuning based on human preferences, they remain vulnerable to adversarial attacks that exploit conflicts between these norms. I argue that this vulnerability reflects a fundamental limitation of existing alignment"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.04679","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.04679/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.04679","created_at":"2026-07-05T11:16:18.798058+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.04679v1","created_at":"2026-07-05T11:16:18.798058+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.04679","created_at":"2026-07-05T11:16:18.798058+00:00"},{"alias_kind":"pith_short_12","alias_value":"ARDP5LAFFKUS","created_at":"2026-07-05T11:16:18.798058+00:00"},{"alias_kind":"pith_short_16","alias_value":"ARDP5LAFFKUSZGXB","created_at":"2026-07-05T11:16:18.798058+00:00"},{"alias_kind":"pith_short_8","alias_value":"ARDP5LAF","created_at":"2026-07-05T11:16:18.798058+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ARDP5LAFFKUSZGXBNPCYJ5DMTT","json":"https://pith.science/pith/ARDP5LAFFKUSZGXBNPCYJ5DMTT.json","graph_json":"https://pith.science/api/pith-number/ARDP5LAFFKUSZGXBNPCYJ5DMTT/graph.json","events_json":"https://pith.science/api/pith-number/ARDP5LAFFKUSZGXBNPCYJ5DMTT/events.json","paper":"https://pith.science/paper/ARDP5LAF"},"agent_actions":{"view_html":"https://pith.science/pith/ARDP5LAFFKUSZGXBNPCYJ5DMTT","download_json":"https://pith.science/pith/ARDP5LAFFKUSZGXBNPCYJ5DMTT.json","view_paper":"https://pith.science/paper/ARDP5LAF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.04679&json=true","fetch_graph":"https://pith.science/api/pith-number/ARDP5LAFFKUSZGXBNPCYJ5DMTT/graph.json","fetch_events":"https://pith.science/api/pith-number/ARDP5LAFFKUSZGXBNPCYJ5DMTT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ARDP5LAFFKUSZGXBNPCYJ5DMTT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ARDP5LAFFKUSZGXBNPCYJ5DMTT/action/storage_attestation","attest_author":"https://pith.science/pith/ARDP5LAFFKUSZGXBNPCYJ5DMTT/action/author_attestation","sign_citation":"https://pith.science/pith/ARDP5LAFFKUSZGXBNPCYJ5DMTT/action/citation_signature","submit_replication":"https://pith.science/pith/ARDP5LAFFKUSZGXBNPCYJ5DMTT/action/replication_record"}},"created_at":"2026-07-05T11:16:18.798058+00:00","updated_at":"2026-07-05T11:16:18.798058+00:00"}