{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KSQGG4DBR224K4LIHLUDNATYJG","short_pith_number":"pith:KSQGG4DB","schema_version":"1.0","canonical_sha256":"54a06370618eb5c571683ae836827849a17c22841fad0cddf4909521de41e500","source":{"kind":"arxiv","id":"2408.04655","version":2},"attestation_state":"computed","paper":{"title":"Strong and weak alignment of large language models with human values","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Marceau Nahon, Mehdi Khamassi, Raja Chatila","submitted_at":"2024-08-05T11:27:51Z","abstract_excerpt":"Minimizing negative impacts of Artificial Intelligent (AI) systems on human societies without human supervision requires them to be able to align with human values. However, most current work only addresses this issue from a technical point of view, e.g., improving current methods relying on reinforcement learning from human feedback, neglecting what it means and is required for alignment to occur. Here, we propose to distinguish strong and weak value alignment. Strong alignment requires cognitive abilities (either human-like or different from humans) such as understanding and reasoning about "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.04655","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-05T11:27:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ef0a9d742f0024419a98d1dc74e5d41d4b8369db9aa2f1e5b5ebf7d62d762d19","abstract_canon_sha256":"970b0837fc73782a36bb6e092f2f8f4d125648675c65c82b5184cd03af16c7f1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:17.309720Z","signature_b64":"rLx97ilDbnBXDDANhGwvs/UJWMzdMmqtXdPJEXerBhgryi8Jatbi7VbRE3RIMKu+60txCIy+QpdDwVglN7gADA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54a06370618eb5c571683ae836827849a17c22841fad0cddf4909521de41e500","last_reissued_at":"2026-07-05T08:54:17.309191Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:17.309191Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Strong and weak alignment of large language models with human values","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Marceau Nahon, Mehdi Khamassi, Raja Chatila","submitted_at":"2024-08-05T11:27:51Z","abstract_excerpt":"Minimizing negative impacts of Artificial Intelligent (AI) systems on human societies without human supervision requires them to be able to align with human values. However, most current work only addresses this issue from a technical point of view, e.g., improving current methods relying on reinforcement learning from human feedback, neglecting what it means and is required for alignment to occur. Here, we propose to distinguish strong and weak value alignment. Strong alignment requires cognitive abilities (either human-like or different from humans) such as understanding and reasoning about "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.04655","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.04655/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.04655","created_at":"2026-07-05T08:54:17.309253+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.04655v2","created_at":"2026-07-05T08:54:17.309253+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.04655","created_at":"2026-07-05T08:54:17.309253+00:00"},{"alias_kind":"pith_short_12","alias_value":"KSQGG4DBR224","created_at":"2026-07-05T08:54:17.309253+00:00"},{"alias_kind":"pith_short_16","alias_value":"KSQGG4DBR224K4LI","created_at":"2026-07-05T08:54:17.309253+00:00"},{"alias_kind":"pith_short_8","alias_value":"KSQGG4DB","created_at":"2026-07-05T08:54:17.309253+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.05583","citing_title":"The Judgment-Consequence Gap: LLM Moral Reasoning in Healthcare Decisions","ref_index":39,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KSQGG4DBR224K4LIHLUDNATYJG","json":"https://pith.science/pith/KSQGG4DBR224K4LIHLUDNATYJG.json","graph_json":"https://pith.science/api/pith-number/KSQGG4DBR224K4LIHLUDNATYJG/graph.json","events_json":"https://pith.science/api/pith-number/KSQGG4DBR224K4LIHLUDNATYJG/events.json","paper":"https://pith.science/paper/KSQGG4DB"},"agent_actions":{"view_html":"https://pith.science/pith/KSQGG4DBR224K4LIHLUDNATYJG","download_json":"https://pith.science/pith/KSQGG4DBR224K4LIHLUDNATYJG.json","view_paper":"https://pith.science/paper/KSQGG4DB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.04655&json=true","fetch_graph":"https://pith.science/api/pith-number/KSQGG4DBR224K4LIHLUDNATYJG/graph.json","fetch_events":"https://pith.science/api/pith-number/KSQGG4DBR224K4LIHLUDNATYJG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KSQGG4DBR224K4LIHLUDNATYJG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KSQGG4DBR224K4LIHLUDNATYJG/action/storage_attestation","attest_author":"https://pith.science/pith/KSQGG4DBR224K4LIHLUDNATYJG/action/author_attestation","sign_citation":"https://pith.science/pith/KSQGG4DBR224K4LIHLUDNATYJG/action/citation_signature","submit_replication":"https://pith.science/pith/KSQGG4DBR224K4LIHLUDNATYJG/action/replication_record"}},"created_at":"2026-07-05T08:54:17.309253+00:00","updated_at":"2026-07-05T08:54:17.309253+00:00"}