{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2ZV6Y7C3Q7E2LVUTBLUXLBINDH","short_pith_number":"pith:2ZV6Y7C3","schema_version":"1.0","canonical_sha256":"d66bec7c5b87c9a5d6930ae975850d19ca2dc15544534beffff9348d69ad6c8f","source":{"kind":"arxiv","id":"2410.07826","version":1},"attestation_state":"computed","paper":{"title":"Fine-Tuning Language Models for Ethical Ambiguity: A Comparative Study of Alignment with Human Responses","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aneesa Maity, Jonathan Lu, Kevin Zhu, Pranav Senthilkumar, Prisha Jain, Visshwa Balasubramanian","submitted_at":"2024-10-10T11:24:04Z","abstract_excerpt":"Language models often misinterpret human intentions due to their handling of ambiguity, a limitation well-recognized in NLP research. While morally clear scenarios are more discernible to LLMs, greater difficulty is encountered in morally ambiguous contexts. In this investigation, we explored LLM calibration to show that human and LLM judgments are poorly aligned in such scenarios. We used two curated datasets from the Scruples project for evaluation: DILEMMAS, which involves pairs of distinct moral scenarios to assess the model's ability to compare and contrast ethical situations, and ANECDOT"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.07826","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-10T11:24:04Z","cross_cats_sorted":[],"title_canon_sha256":"675d61e87c61a50bbc60cc910cdca78afdb99bec1b828aeb1b282d755bbec510","abstract_canon_sha256":"70144431a4382f127adefc8dd9518a6ad067f7e83bbff369475ca84263359f84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:18:44.740253Z","signature_b64":"0ME2IOCKePwP9M/4xMxMhNOhvQv29a59sBzJ/4NplRlumIsmjMvDD6sKhYjzjA5uVbuuLdgDU1U3eWRvbCa1DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d66bec7c5b87c9a5d6930ae975850d19ca2dc15544534beffff9348d69ad6c8f","last_reissued_at":"2026-07-05T09:18:44.739769Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:18:44.739769Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fine-Tuning Language Models for Ethical Ambiguity: A Comparative Study of Alignment with Human Responses","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aneesa Maity, Jonathan Lu, Kevin Zhu, Pranav Senthilkumar, Prisha Jain, Visshwa Balasubramanian","submitted_at":"2024-10-10T11:24:04Z","abstract_excerpt":"Language models often misinterpret human intentions due to their handling of ambiguity, a limitation well-recognized in NLP research. While morally clear scenarios are more discernible to LLMs, greater difficulty is encountered in morally ambiguous contexts. In this investigation, we explored LLM calibration to show that human and LLM judgments are poorly aligned in such scenarios. We used two curated datasets from the Scruples project for evaluation: DILEMMAS, which involves pairs of distinct moral scenarios to assess the model's ability to compare and contrast ethical situations, and ANECDOT"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.07826","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.07826/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.07826","created_at":"2026-07-05T09:18:44.739833+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.07826v1","created_at":"2026-07-05T09:18:44.739833+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.07826","created_at":"2026-07-05T09:18:44.739833+00:00"},{"alias_kind":"pith_short_12","alias_value":"2ZV6Y7C3Q7E2","created_at":"2026-07-05T09:18:44.739833+00:00"},{"alias_kind":"pith_short_16","alias_value":"2ZV6Y7C3Q7E2LVUT","created_at":"2026-07-05T09:18:44.739833+00:00"},{"alias_kind":"pith_short_8","alias_value":"2ZV6Y7C3","created_at":"2026-07-05T09:18:44.739833+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.13952","citing_title":"The Dual-use Dilemma in LLMs: Do Empowering Ethical Capacities Make a Degraded Utility?","ref_index":30,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2ZV6Y7C3Q7E2LVUTBLUXLBINDH","json":"https://pith.science/pith/2ZV6Y7C3Q7E2LVUTBLUXLBINDH.json","graph_json":"https://pith.science/api/pith-number/2ZV6Y7C3Q7E2LVUTBLUXLBINDH/graph.json","events_json":"https://pith.science/api/pith-number/2ZV6Y7C3Q7E2LVUTBLUXLBINDH/events.json","paper":"https://pith.science/paper/2ZV6Y7C3"},"agent_actions":{"view_html":"https://pith.science/pith/2ZV6Y7C3Q7E2LVUTBLUXLBINDH","download_json":"https://pith.science/pith/2ZV6Y7C3Q7E2LVUTBLUXLBINDH.json","view_paper":"https://pith.science/paper/2ZV6Y7C3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.07826&json=true","fetch_graph":"https://pith.science/api/pith-number/2ZV6Y7C3Q7E2LVUTBLUXLBINDH/graph.json","fetch_events":"https://pith.science/api/pith-number/2ZV6Y7C3Q7E2LVUTBLUXLBINDH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2ZV6Y7C3Q7E2LVUTBLUXLBINDH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2ZV6Y7C3Q7E2LVUTBLUXLBINDH/action/storage_attestation","attest_author":"https://pith.science/pith/2ZV6Y7C3Q7E2LVUTBLUXLBINDH/action/author_attestation","sign_citation":"https://pith.science/pith/2ZV6Y7C3Q7E2LVUTBLUXLBINDH/action/citation_signature","submit_replication":"https://pith.science/pith/2ZV6Y7C3Q7E2LVUTBLUXLBINDH/action/replication_record"}},"created_at":"2026-07-05T09:18:44.739833+00:00","updated_at":"2026-07-05T09:18:44.739833+00:00"}