{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KDK42KR5P7GH4UQGFSWKHMJFDP","short_pith_number":"pith:KDK42KR5","schema_version":"1.0","canonical_sha256":"50d5cd2a3d7fcc7e52062caca3b1251be56859e1677e984937dcb5868d415611","source":{"kind":"arxiv","id":"2312.06281","version":2},"attestation_state":"computed","paper":{"title":"EQ-Bench: An Emotional Intelligence Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Samuel J. Paech","submitted_at":"2023-12-11T10:35:32Z","abstract_excerpt":"We introduce EQ-Bench, a novel benchmark designed to evaluate aspects of emotional intelligence in Large Language Models (LLMs). We assess the ability of LLMs to understand complex emotions and social interactions by asking them to predict the intensity of emotional states of characters in a dialogue. The benchmark is able to discriminate effectively between a wide range of models. We find that EQ-Bench correlates strongly with comprehensive multi-domain benchmarks like MMLU (Hendrycks et al., 2020) (r=0.97), indicating that we may be capturing similar aspects of broad intelligence. Our benchm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.06281","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-12-11T10:35:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"75f64f7d5165d265abcfa2ec1b16e5ae0537f66cc30ba32fc13a8c9897289c39","abstract_canon_sha256":"dd2f48d156ae4f393b744fb5c8335fbe245b138e4d1b32ca42f0464e4627710d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:29:54.984027Z","signature_b64":"E8k85dZuJpmrtZ/jsfvtUQe0J+gFxN9OGm7umkZUv8Va9nyBh7Ap1DVCNL7xijRZKcPSPcIplmwq7pW+kmR3Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50d5cd2a3d7fcc7e52062caca3b1251be56859e1677e984937dcb5868d415611","last_reissued_at":"2026-07-05T07:29:54.983613Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:29:54.983613Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EQ-Bench: An Emotional Intelligence Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Samuel J. Paech","submitted_at":"2023-12-11T10:35:32Z","abstract_excerpt":"We introduce EQ-Bench, a novel benchmark designed to evaluate aspects of emotional intelligence in Large Language Models (LLMs). We assess the ability of LLMs to understand complex emotions and social interactions by asking them to predict the intensity of emotional states of characters in a dialogue. The benchmark is able to discriminate effectively between a wide range of models. We find that EQ-Bench correlates strongly with comprehensive multi-domain benchmarks like MMLU (Hendrycks et al., 2020) (r=0.97), indicating that we may be capturing similar aspects of broad intelligence. Our benchm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.06281","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.06281/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.06281","created_at":"2026-07-05T07:29:54.983667+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.06281v2","created_at":"2026-07-05T07:29:54.983667+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.06281","created_at":"2026-07-05T07:29:54.983667+00:00"},{"alias_kind":"pith_short_12","alias_value":"KDK42KR5P7GH","created_at":"2026-07-05T07:29:54.983667+00:00"},{"alias_kind":"pith_short_16","alias_value":"KDK42KR5P7GH4UQG","created_at":"2026-07-05T07:29:54.983667+00:00"},{"alias_kind":"pith_short_8","alias_value":"KDK42KR5","created_at":"2026-07-05T07:29:54.983667+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.21739","citing_title":"AttuneBench: A Conversation-Based Benchmark for LLM Emotional Intelligence","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24686","citing_title":"Emotional intelligence in large language models is fragmented across perception, cognition, and interaction","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26579","citing_title":"Focal Reward: Balanced Reinforcement Learning under Rubric-Based Rewards","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23651","citing_title":"How Human-Like Are Large Language Models? A Register-Aware Linguistic Evaluation Framework","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21739","citing_title":"AttuneBench: A Conversation-Based Benchmark for LLM Emotional Intelligence","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.10746","citing_title":"RECAP: Transparent Inference-Time Emotion Alignment for Medical Dialogue Systems","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14234","citing_title":"Compute as Teacher: Turning Inference Compute Into Reference-Free Supervision","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06194","citing_title":"MICA: Multi-granularity Intertemporal Credit Assignment for Long-Horizon Emotional Support Dialogue","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2603.26680","citing_title":"AlpsBench: An LLM Personalization Benchmark for Real-Dialogue Memorization and Preference Alignment","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04831","citing_title":"StoryAlign: Evaluating and Training Reward Models for Story Generation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09854","citing_title":"Spoiler Alert: Narrative Forecasting as a Metric for Tension in LLM Storytelling","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KDK42KR5P7GH4UQGFSWKHMJFDP","json":"https://pith.science/pith/KDK42KR5P7GH4UQGFSWKHMJFDP.json","graph_json":"https://pith.science/api/pith-number/KDK42KR5P7GH4UQGFSWKHMJFDP/graph.json","events_json":"https://pith.science/api/pith-number/KDK42KR5P7GH4UQGFSWKHMJFDP/events.json","paper":"https://pith.science/paper/KDK42KR5"},"agent_actions":{"view_html":"https://pith.science/pith/KDK42KR5P7GH4UQGFSWKHMJFDP","download_json":"https://pith.science/pith/KDK42KR5P7GH4UQGFSWKHMJFDP.json","view_paper":"https://pith.science/paper/KDK42KR5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.06281&json=true","fetch_graph":"https://pith.science/api/pith-number/KDK42KR5P7GH4UQGFSWKHMJFDP/graph.json","fetch_events":"https://pith.science/api/pith-number/KDK42KR5P7GH4UQGFSWKHMJFDP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KDK42KR5P7GH4UQGFSWKHMJFDP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KDK42KR5P7GH4UQGFSWKHMJFDP/action/storage_attestation","attest_author":"https://pith.science/pith/KDK42KR5P7GH4UQGFSWKHMJFDP/action/author_attestation","sign_citation":"https://pith.science/pith/KDK42KR5P7GH4UQGFSWKHMJFDP/action/citation_signature","submit_replication":"https://pith.science/pith/KDK42KR5P7GH4UQGFSWKHMJFDP/action/replication_record"}},"created_at":"2026-07-05T07:29:54.983667+00:00","updated_at":"2026-07-05T07:29:54.983667+00:00"}