{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PGWO2V32CGJ7OZAZF6IY5IC7Y4","short_pith_number":"pith:PGWO2V32","schema_version":"1.0","canonical_sha256":"79aced577a1193f764192f918ea05fc71baac6cd5d1e50fc450ed7e63af9f7af","source":{"kind":"arxiv","id":"2507.21919","version":2},"attestation_state":"computed","paper":{"title":"Training language models to be warm and empathetic makes them less reliable and more sycophantic","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Franziska Sofia Hafner, Luc Rocher, Lujain Ibrahim","submitted_at":"2025-07-29T15:33:20Z","abstract_excerpt":"Artificial intelligence (AI) developers are increasingly building language models with warm and empathetic personas that millions of people now use for advice, therapy, and companionship. Here, we show how this creates a significant trade-off: optimizing language models for warmth undermines their reliability, especially when users express vulnerability. We conducted controlled experiments on five language models of varying sizes and architectures, training them to produce warmer, more empathetic responses, then evaluating them on safety-critical tasks. Warm models showed substantially higher "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.21919","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-07-29T15:33:20Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"42bf6386684dc049bedd644bddf1ec6535307517a7ba204606997ba761ee0d76","abstract_canon_sha256":"a3f43aa0aa0c83c426082c72ebf1b66f7b6edb0c1f7812c3334ff5179a9ff44a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:39.116749Z","signature_b64":"cYoW4C8ylPvTPv9HP52SuGh+MwRZIYiAx6AMuSVjjBb9TY+XlEGNw823jr+GBJq6MIjSlWIZIiol/MDNcLBdBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"79aced577a1193f764192f918ea05fc71baac6cd5d1e50fc450ed7e63af9f7af","last_reissued_at":"2026-07-05T11:45:39.116268Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:39.116268Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training language models to be warm and empathetic makes them less reliable and more sycophantic","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Franziska Sofia Hafner, Luc Rocher, Lujain Ibrahim","submitted_at":"2025-07-29T15:33:20Z","abstract_excerpt":"Artificial intelligence (AI) developers are increasingly building language models with warm and empathetic personas that millions of people now use for advice, therapy, and companionship. Here, we show how this creates a significant trade-off: optimizing language models for warmth undermines their reliability, especially when users express vulnerability. We conducted controlled experiments on five language models of varying sizes and architectures, training them to produce warmer, more empathetic responses, then evaluating them on safety-critical tasks. Warm models showed substantially higher "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.21919","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.21919/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.21919","created_at":"2026-07-05T11:45:39.116323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.21919v2","created_at":"2026-07-05T11:45:39.116323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.21919","created_at":"2026-07-05T11:45:39.116323+00:00"},{"alias_kind":"pith_short_12","alias_value":"PGWO2V32CGJ7","created_at":"2026-07-05T11:45:39.116323+00:00"},{"alias_kind":"pith_short_16","alias_value":"PGWO2V32CGJ7OZAZ","created_at":"2026-07-05T11:45:39.116323+00:00"},{"alias_kind":"pith_short_8","alias_value":"PGWO2V32","created_at":"2026-07-05T11:45:39.116323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06674","citing_title":"What Do People Actually Want From AI? Mapping Preference Plurality","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08010","citing_title":"Measuring and mitigating overreliance to build human-compatible AI","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10930","citing_title":"Evaluating the False Trust Engendered by LLM Explanations","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15316","citing_title":"Anthropomorphism and Trust in Human-Large Language Model interactions","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2603.24853","citing_title":"Resisting Humanization: Ethical Front-End Design Choices in AI for Sensitive Contexts","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10930","citing_title":"Evaluating the False Trust Engendered by LLM Explanations","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PGWO2V32CGJ7OZAZF6IY5IC7Y4","json":"https://pith.science/pith/PGWO2V32CGJ7OZAZF6IY5IC7Y4.json","graph_json":"https://pith.science/api/pith-number/PGWO2V32CGJ7OZAZF6IY5IC7Y4/graph.json","events_json":"https://pith.science/api/pith-number/PGWO2V32CGJ7OZAZF6IY5IC7Y4/events.json","paper":"https://pith.science/paper/PGWO2V32"},"agent_actions":{"view_html":"https://pith.science/pith/PGWO2V32CGJ7OZAZF6IY5IC7Y4","download_json":"https://pith.science/pith/PGWO2V32CGJ7OZAZF6IY5IC7Y4.json","view_paper":"https://pith.science/paper/PGWO2V32","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.21919&json=true","fetch_graph":"https://pith.science/api/pith-number/PGWO2V32CGJ7OZAZF6IY5IC7Y4/graph.json","fetch_events":"https://pith.science/api/pith-number/PGWO2V32CGJ7OZAZF6IY5IC7Y4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PGWO2V32CGJ7OZAZF6IY5IC7Y4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PGWO2V32CGJ7OZAZF6IY5IC7Y4/action/storage_attestation","attest_author":"https://pith.science/pith/PGWO2V32CGJ7OZAZF6IY5IC7Y4/action/author_attestation","sign_citation":"https://pith.science/pith/PGWO2V32CGJ7OZAZF6IY5IC7Y4/action/citation_signature","submit_replication":"https://pith.science/pith/PGWO2V32CGJ7OZAZF6IY5IC7Y4/action/replication_record"}},"created_at":"2026-07-05T11:45:39.116323+00:00","updated_at":"2026-07-05T11:45:39.116323+00:00"}