{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7N6T32U7DJESTGIIF4WAWEG46C","short_pith_number":"pith:7N6T32U7","schema_version":"1.0","canonical_sha256":"fb7d3dea9f1a492999082f2c0b10dcf0aa5e16c5b38a80a996ee3150ebd5a2b1","source":{"kind":"arxiv","id":"2505.08106","version":1},"attestation_state":"computed","paper":{"title":"Are LLMs complicated ethical dilemma analyzers?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Allen Liu, Jesse Yao, Jiashen (Jason) Du, Zhekai Zhang","submitted_at":"2025-05-12T22:35:07Z","abstract_excerpt":"One open question in the study of Large Language Models (LLMs) is whether they can emulate human ethical reasoning and act as believable proxies for human judgment. To investigate this, we introduce a benchmark dataset comprising 196 real-world ethical dilemmas and expert opinions, each segmented into five structured components: Introduction, Key Factors, Historical Theoretical Perspectives, Resolution Strategies, and Key Takeaways. We also collect non-expert human responses for comparison, limited to the Key Factors section due to their brevity. We evaluate multiple frontier LLMs (GPT-4o-mini"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.08106","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-12T22:35:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0070d4ddec3a6e5f97c109285a365f9689d192b9136a5c2bf44689cb0c8df851","abstract_canon_sha256":"81ef5c0c95a72295bdeacf9d3f0e56ae4186cb8caa4adc3e3024fa632f403cb0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:02.714172Z","signature_b64":"Q4NC6abJI7DfcFrLp5e7fZI4EFdoXt8SYW4tmbGAfKIcV4e1RDyd8ILDEOGo9gU49kWbp9ObgWNHljjnB0QHAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb7d3dea9f1a492999082f2c0b10dcf0aa5e16c5b38a80a996ee3150ebd5a2b1","last_reissued_at":"2026-07-05T11:02:02.713756Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:02.713756Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are LLMs complicated ethical dilemma analyzers?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Allen Liu, Jesse Yao, Jiashen (Jason) Du, Zhekai Zhang","submitted_at":"2025-05-12T22:35:07Z","abstract_excerpt":"One open question in the study of Large Language Models (LLMs) is whether they can emulate human ethical reasoning and act as believable proxies for human judgment. To investigate this, we introduce a benchmark dataset comprising 196 real-world ethical dilemmas and expert opinions, each segmented into five structured components: Introduction, Key Factors, Historical Theoretical Perspectives, Resolution Strategies, and Key Takeaways. We also collect non-expert human responses for comparison, limited to the Key Factors section due to their brevity. We evaluate multiple frontier LLMs (GPT-4o-mini"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08106","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.08106/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.08106","created_at":"2026-07-05T11:02:02.713811+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.08106v1","created_at":"2026-07-05T11:02:02.713811+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08106","created_at":"2026-07-05T11:02:02.713811+00:00"},{"alias_kind":"pith_short_12","alias_value":"7N6T32U7DJES","created_at":"2026-07-05T11:02:02.713811+00:00"},{"alias_kind":"pith_short_16","alias_value":"7N6T32U7DJESTGII","created_at":"2026-07-05T11:02:02.713811+00:00"},{"alias_kind":"pith_short_8","alias_value":"7N6T32U7","created_at":"2026-07-05T11:02:02.713811+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7N6T32U7DJESTGIIF4WAWEG46C","json":"https://pith.science/pith/7N6T32U7DJESTGIIF4WAWEG46C.json","graph_json":"https://pith.science/api/pith-number/7N6T32U7DJESTGIIF4WAWEG46C/graph.json","events_json":"https://pith.science/api/pith-number/7N6T32U7DJESTGIIF4WAWEG46C/events.json","paper":"https://pith.science/paper/7N6T32U7"},"agent_actions":{"view_html":"https://pith.science/pith/7N6T32U7DJESTGIIF4WAWEG46C","download_json":"https://pith.science/pith/7N6T32U7DJESTGIIF4WAWEG46C.json","view_paper":"https://pith.science/paper/7N6T32U7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.08106&json=true","fetch_graph":"https://pith.science/api/pith-number/7N6T32U7DJESTGIIF4WAWEG46C/graph.json","fetch_events":"https://pith.science/api/pith-number/7N6T32U7DJESTGIIF4WAWEG46C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7N6T32U7DJESTGIIF4WAWEG46C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7N6T32U7DJESTGIIF4WAWEG46C/action/storage_attestation","attest_author":"https://pith.science/pith/7N6T32U7DJESTGIIF4WAWEG46C/action/author_attestation","sign_citation":"https://pith.science/pith/7N6T32U7DJESTGIIF4WAWEG46C/action/citation_signature","submit_replication":"https://pith.science/pith/7N6T32U7DJESTGIIF4WAWEG46C/action/replication_record"}},"created_at":"2026-07-05T11:02:02.713811+00:00","updated_at":"2026-07-05T11:02:02.713811+00:00"}