{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SWZEYATZF765RTCS5N4IHTLN7L","short_pith_number":"pith:SWZEYATZ","schema_version":"1.0","canonical_sha256":"95b24c02792ffdd8cc52eb7883cd6dfaec8fc2d5be46e32d817f7244da381f3e","source":{"kind":"arxiv","id":"2506.14948","version":1},"attestation_state":"computed","paper":{"title":"Structured Moral Reasoning in Language Models: A Value-Grounded Evaluation Framework","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.HC","authors_text":"David Jurgens, Lu Wang, Mohna Chakraborty","submitted_at":"2025-06-17T19:59:44Z","abstract_excerpt":"Large language models (LLMs) are increasingly deployed in domains requiring moral understanding, yet their reasoning often remains shallow, and misaligned with human reasoning. Unlike humans, whose moral reasoning integrates contextual trade-offs, value systems, and ethical theories, LLMs often rely on surface patterns, leading to biased decisions in morally and ethically complex scenarios. To address this gap, we present a value-grounded framework for evaluating and distilling structured moral reasoning in LLMs. We benchmark 12 open-source models across four moral datasets using a taxonomy of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.14948","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.HC","submitted_at":"2025-06-17T19:59:44Z","cross_cats_sorted":[],"title_canon_sha256":"78d13bc2e69be5c19bd207f8c3c1ca8c5a16bb973065b4ca1ba3940586c573e6","abstract_canon_sha256":"356746f943572124a58ac6c8515a86dfd0cce0a06619ef8af04245408341c35c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:34.602879Z","signature_b64":"GN2CqWVax3i7nNeLa5q72tS4llF3/kry873PB5j+JJ1H6uf5IHQzgPg9dDMKiPp+xVU90BFWd6dc2LAzGFXKDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"95b24c02792ffdd8cc52eb7883cd6dfaec8fc2d5be46e32d817f7244da381f3e","last_reissued_at":"2026-07-05T11:23:34.602377Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:34.602377Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Structured Moral Reasoning in Language Models: A Value-Grounded Evaluation Framework","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.HC","authors_text":"David Jurgens, Lu Wang, Mohna Chakraborty","submitted_at":"2025-06-17T19:59:44Z","abstract_excerpt":"Large language models (LLMs) are increasingly deployed in domains requiring moral understanding, yet their reasoning often remains shallow, and misaligned with human reasoning. Unlike humans, whose moral reasoning integrates contextual trade-offs, value systems, and ethical theories, LLMs often rely on surface patterns, leading to biased decisions in morally and ethically complex scenarios. To address this gap, we present a value-grounded framework for evaluating and distilling structured moral reasoning in LLMs. We benchmark 12 open-source models across four moral datasets using a taxonomy of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.14948","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.14948/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.14948","created_at":"2026-07-05T11:23:34.602440+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.14948v1","created_at":"2026-07-05T11:23:34.602440+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.14948","created_at":"2026-07-05T11:23:34.602440+00:00"},{"alias_kind":"pith_short_12","alias_value":"SWZEYATZF765","created_at":"2026-07-05T11:23:34.602440+00:00"},{"alias_kind":"pith_short_16","alias_value":"SWZEYATZF765RTCS","created_at":"2026-07-05T11:23:34.602440+00:00"},{"alias_kind":"pith_short_8","alias_value":"SWZEYATZ","created_at":"2026-07-05T11:23:34.602440+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.03609","citing_title":"Where Paths Split: Localized, Calibrated Control of Moral Reasoning in Large Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00270","citing_title":"Are You the A-hole? A Fair, Multi-Perspective Ethical Reasoning Framework","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SWZEYATZF765RTCS5N4IHTLN7L","json":"https://pith.science/pith/SWZEYATZF765RTCS5N4IHTLN7L.json","graph_json":"https://pith.science/api/pith-number/SWZEYATZF765RTCS5N4IHTLN7L/graph.json","events_json":"https://pith.science/api/pith-number/SWZEYATZF765RTCS5N4IHTLN7L/events.json","paper":"https://pith.science/paper/SWZEYATZ"},"agent_actions":{"view_html":"https://pith.science/pith/SWZEYATZF765RTCS5N4IHTLN7L","download_json":"https://pith.science/pith/SWZEYATZF765RTCS5N4IHTLN7L.json","view_paper":"https://pith.science/paper/SWZEYATZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.14948&json=true","fetch_graph":"https://pith.science/api/pith-number/SWZEYATZF765RTCS5N4IHTLN7L/graph.json","fetch_events":"https://pith.science/api/pith-number/SWZEYATZF765RTCS5N4IHTLN7L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SWZEYATZF765RTCS5N4IHTLN7L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SWZEYATZF765RTCS5N4IHTLN7L/action/storage_attestation","attest_author":"https://pith.science/pith/SWZEYATZF765RTCS5N4IHTLN7L/action/author_attestation","sign_citation":"https://pith.science/pith/SWZEYATZF765RTCS5N4IHTLN7L/action/citation_signature","submit_replication":"https://pith.science/pith/SWZEYATZF765RTCS5N4IHTLN7L/action/replication_record"}},"created_at":"2026-07-05T11:23:34.602440+00:00","updated_at":"2026-07-05T11:23:34.602440+00:00"}