{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CEXM45RACWWXNHXBXOV5EM2AE7","short_pith_number":"pith:CEXM45RA","schema_version":"1.0","canonical_sha256":"112ece762015ad769ee1bbabd2334027c2b430cf36806be415425e3a4565f869","source":{"kind":"arxiv","id":"2501.18081","version":1},"attestation_state":"computed","paper":{"title":"Normative Evaluation of Large Language Models with Everyday Moral Dilemmas","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.AI","authors_text":"Pratik S. Sachdeva, Tom van Nuenen","submitted_at":"2025-01-30T01:29:46Z","abstract_excerpt":"The rapid adoption of large language models (LLMs) has spurred extensive research into their encoded moral norms and decision-making processes. Much of this research relies on prompting LLMs with survey-style questions to assess how well models are aligned with certain demographic groups, moral beliefs, or political ideologies. While informative, the adherence of these approaches to relatively superficial constructs tends to oversimplify the complexity and nuance underlying everyday moral dilemmas. We argue that auditing LLMs along more detailed axes of human interaction is of paramount import"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.18081","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-01-30T01:29:46Z","cross_cats_sorted":["cs.CY"],"title_canon_sha256":"bd529de4632fc45fa57579b47afc6a8f6d2081b4b89aeadc3467482cbb33bffd","abstract_canon_sha256":"abfbc8c9848454aa583ef2512eca51617bc0204c1e0268a77c1ec21a0a42cb85"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:07:06.768552Z","signature_b64":"8xYJtOmJus4DkPz0cxTam5tVNTKHQFLyTKQTi7KfiRL3TNI5EDLY+x+m1nOXUchvlYir1dYvr4YwpY6HHwhWAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"112ece762015ad769ee1bbabd2334027c2b430cf36806be415425e3a4565f869","last_reissued_at":"2026-07-05T10:07:06.768016Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:07:06.768016Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Normative Evaluation of Large Language Models with Everyday Moral Dilemmas","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.AI","authors_text":"Pratik S. Sachdeva, Tom van Nuenen","submitted_at":"2025-01-30T01:29:46Z","abstract_excerpt":"The rapid adoption of large language models (LLMs) has spurred extensive research into their encoded moral norms and decision-making processes. Much of this research relies on prompting LLMs with survey-style questions to assess how well models are aligned with certain demographic groups, moral beliefs, or political ideologies. While informative, the adherence of these approaches to relatively superficial constructs tends to oversimplify the complexity and nuance underlying everyday moral dilemmas. We argue that auditing LLMs along more detailed axes of human interaction is of paramount import"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.18081","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.18081/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.18081","created_at":"2026-07-05T10:07:06.768073+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.18081v1","created_at":"2026-07-05T10:07:06.768073+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.18081","created_at":"2026-07-05T10:07:06.768073+00:00"},{"alias_kind":"pith_short_12","alias_value":"CEXM45RACWWX","created_at":"2026-07-05T10:07:06.768073+00:00"},{"alias_kind":"pith_short_16","alias_value":"CEXM45RACWWXNHXB","created_at":"2026-07-05T10:07:06.768073+00:00"},{"alias_kind":"pith_short_8","alias_value":"CEXM45RA","created_at":"2026-07-05T10:07:06.768073+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.17216","citing_title":"The Pluralistic Moral Gap: Understanding Judgment and Value Differences between Humans and Large Language Models","ref_index":33,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CEXM45RACWWXNHXBXOV5EM2AE7","json":"https://pith.science/pith/CEXM45RACWWXNHXBXOV5EM2AE7.json","graph_json":"https://pith.science/api/pith-number/CEXM45RACWWXNHXBXOV5EM2AE7/graph.json","events_json":"https://pith.science/api/pith-number/CEXM45RACWWXNHXBXOV5EM2AE7/events.json","paper":"https://pith.science/paper/CEXM45RA"},"agent_actions":{"view_html":"https://pith.science/pith/CEXM45RACWWXNHXBXOV5EM2AE7","download_json":"https://pith.science/pith/CEXM45RACWWXNHXBXOV5EM2AE7.json","view_paper":"https://pith.science/paper/CEXM45RA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.18081&json=true","fetch_graph":"https://pith.science/api/pith-number/CEXM45RACWWXNHXBXOV5EM2AE7/graph.json","fetch_events":"https://pith.science/api/pith-number/CEXM45RACWWXNHXBXOV5EM2AE7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CEXM45RACWWXNHXBXOV5EM2AE7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CEXM45RACWWXNHXBXOV5EM2AE7/action/storage_attestation","attest_author":"https://pith.science/pith/CEXM45RACWWXNHXBXOV5EM2AE7/action/author_attestation","sign_citation":"https://pith.science/pith/CEXM45RACWWXNHXBXOV5EM2AE7/action/citation_signature","submit_replication":"https://pith.science/pith/CEXM45RACWWXNHXBXOV5EM2AE7/action/replication_record"}},"created_at":"2026-07-05T10:07:06.768073+00:00","updated_at":"2026-07-05T10:07:06.768073+00:00"}