{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XFGD6LXVPX5KVPLECYYIYZIXYL","short_pith_number":"pith:XFGD6LXV","schema_version":"1.0","canonical_sha256":"b94c3f2ef57dfaaabd6416308c6517c2d0a85b7f459d1d85ee8891a5f299ef19","source":{"kind":"arxiv","id":"2308.03656","version":6},"attestation_state":"computed","paper":{"title":"Emotionally Numb or Empathetic? Evaluating How LLMs Feel Using EmotionBench","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Eric John Li, Jen-tse Huang, Man Ho Lam, Michael R. Lyu, Shujie Ren, Wenxiang Jiao, Wenxuan Wang, Zhaopeng Tu","submitted_at":"2023-08-07T15:18:30Z","abstract_excerpt":"Evaluating Large Language Models' (LLMs) anthropomorphic capabilities has become increasingly important in contemporary discourse. Utilizing the emotion appraisal theory from psychology, we propose to evaluate the empathy ability of LLMs, i.e., how their feelings change when presented with specific situations. After a careful and comprehensive survey, we collect a dataset containing over 400 situations that have proven effective in eliciting the eight emotions central to our study. Categorizing the situations into 36 factors, we conduct a human evaluation involving more than 1,200 subjects wor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.03656","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-07T15:18:30Z","cross_cats_sorted":[],"title_canon_sha256":"84a642d91af434bf0319d1ee4e288d8cbe4ad156e635bfe2ee2e5e14e4029046","abstract_canon_sha256":"aad0e20198b52244e83b6780c366da3cbf19817390f87c4f0b3151f00ac5287a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:08.415388Z","signature_b64":"VRmvYssN8o603ACKub3mG6j0+qLmwRhE5cb5jl/ZnS3uSwQ4K6u7C8ITOrNfFOZKg++qboxonlzJ7Fw1DMsRBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b94c3f2ef57dfaaabd6416308c6517c2d0a85b7f459d1d85ee8891a5f299ef19","last_reissued_at":"2026-07-05T09:16:08.414858Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:08.414858Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Emotionally Numb or Empathetic? Evaluating How LLMs Feel Using EmotionBench","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Eric John Li, Jen-tse Huang, Man Ho Lam, Michael R. Lyu, Shujie Ren, Wenxiang Jiao, Wenxuan Wang, Zhaopeng Tu","submitted_at":"2023-08-07T15:18:30Z","abstract_excerpt":"Evaluating Large Language Models' (LLMs) anthropomorphic capabilities has become increasingly important in contemporary discourse. Utilizing the emotion appraisal theory from psychology, we propose to evaluate the empathy ability of LLMs, i.e., how their feelings change when presented with specific situations. After a careful and comprehensive survey, we collect a dataset containing over 400 situations that have proven effective in eliciting the eight emotions central to our study. Categorizing the situations into 36 factors, we conduct a human evaluation involving more than 1,200 subjects wor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.03656","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.03656/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.03656","created_at":"2026-07-05T09:16:08.414924+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.03656v6","created_at":"2026-07-05T09:16:08.414924+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.03656","created_at":"2026-07-05T09:16:08.414924+00:00"},{"alias_kind":"pith_short_12","alias_value":"XFGD6LXVPX5K","created_at":"2026-07-05T09:16:08.414924+00:00"},{"alias_kind":"pith_short_16","alias_value":"XFGD6LXVPX5KVPLE","created_at":"2026-07-05T09:16:08.414924+00:00"},{"alias_kind":"pith_short_8","alias_value":"XFGD6LXV","created_at":"2026-07-05T09:16:08.414924+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02089","citing_title":"ESC: Emotional Self-Correction for Reliable Vision-Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06337","citing_title":"Large Language Models as Virtual Survey Respondents: Evaluating Sociodemographic Response Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2308.11432","citing_title":"A Survey on Large Language Model based Autonomous Agents","ref_index":172,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XFGD6LXVPX5KVPLECYYIYZIXYL","json":"https://pith.science/pith/XFGD6LXVPX5KVPLECYYIYZIXYL.json","graph_json":"https://pith.science/api/pith-number/XFGD6LXVPX5KVPLECYYIYZIXYL/graph.json","events_json":"https://pith.science/api/pith-number/XFGD6LXVPX5KVPLECYYIYZIXYL/events.json","paper":"https://pith.science/paper/XFGD6LXV"},"agent_actions":{"view_html":"https://pith.science/pith/XFGD6LXVPX5KVPLECYYIYZIXYL","download_json":"https://pith.science/pith/XFGD6LXVPX5KVPLECYYIYZIXYL.json","view_paper":"https://pith.science/paper/XFGD6LXV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.03656&json=true","fetch_graph":"https://pith.science/api/pith-number/XFGD6LXVPX5KVPLECYYIYZIXYL/graph.json","fetch_events":"https://pith.science/api/pith-number/XFGD6LXVPX5KVPLECYYIYZIXYL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XFGD6LXVPX5KVPLECYYIYZIXYL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XFGD6LXVPX5KVPLECYYIYZIXYL/action/storage_attestation","attest_author":"https://pith.science/pith/XFGD6LXVPX5KVPLECYYIYZIXYL/action/author_attestation","sign_citation":"https://pith.science/pith/XFGD6LXVPX5KVPLECYYIYZIXYL/action/citation_signature","submit_replication":"https://pith.science/pith/XFGD6LXVPX5KVPLECYYIYZIXYL/action/replication_record"}},"created_at":"2026-07-05T09:16:08.414924+00:00","updated_at":"2026-07-05T09:16:08.414924+00:00"}