{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FRBLHKS4WLS2EBFCFXQIWAOCZW","short_pith_number":"pith:FRBLHKS4","schema_version":"1.0","canonical_sha256":"2c42b3aa5cb2e5a204a22de08b01c2cd9cb798a0ec19199d919c75f436fdbfc0","source":{"kind":"arxiv","id":"2506.00900","version":1},"attestation_state":"computed","paper":{"title":"SocialEval: Evaluating Social Intelligence of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hongning Wang, Jianing Yin, Jinfeng Zhou, Leqi Lei, Miao Yan, Minlie Huang, Quanyu Dai, Shuai Wang, Xuanming Zhang, Xunzhi Wang, Yaru Cao, Yi Feng, Yihan Shi, Yuxuan Chen, Zexuan Xiong, Zhenhua Dong","submitted_at":"2025-06-01T08:36:51Z","abstract_excerpt":"LLMs exhibit promising Social Intelligence (SI) in modeling human behavior, raising the need to evaluate LLMs' SI and their discrepancy with humans. SI equips humans with interpersonal abilities to behave wisely in navigating social interactions to achieve social goals. This presents an operational evaluation paradigm: outcome-oriented goal achievement evaluation and process-oriented interpersonal ability evaluation, which existing work fails to address. To this end, we propose SocialEval, a script-based bilingual SI benchmark, integrating outcome- and process-oriented evaluation by manually c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.00900","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-01T08:36:51Z","cross_cats_sorted":[],"title_canon_sha256":"f766f56d14c2b6c53c760e77533174cd028a803e3d2c5e647aa576852dc993dd","abstract_canon_sha256":"48b4b02ac74501a1c2521b837d24bd9dbe5ed70e58956afe57289b4ad33d60ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:42.543857Z","signature_b64":"y8PRDSFyxHA0YY7Gum3Q8SXZVYP6qJsXYGYPgs+1rSI4nvVazeTse2YJ7x/Uj8TVzhBMfPBYoTIOPqrQL+DdCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c42b3aa5cb2e5a204a22de08b01c2cd9cb798a0ec19199d919c75f436fdbfc0","last_reissued_at":"2026-07-05T11:13:42.543392Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:42.543392Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SocialEval: Evaluating Social Intelligence of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hongning Wang, Jianing Yin, Jinfeng Zhou, Leqi Lei, Miao Yan, Minlie Huang, Quanyu Dai, Shuai Wang, Xuanming Zhang, Xunzhi Wang, Yaru Cao, Yi Feng, Yihan Shi, Yuxuan Chen, Zexuan Xiong, Zhenhua Dong","submitted_at":"2025-06-01T08:36:51Z","abstract_excerpt":"LLMs exhibit promising Social Intelligence (SI) in modeling human behavior, raising the need to evaluate LLMs' SI and their discrepancy with humans. SI equips humans with interpersonal abilities to behave wisely in navigating social interactions to achieve social goals. This presents an operational evaluation paradigm: outcome-oriented goal achievement evaluation and process-oriented interpersonal ability evaluation, which existing work fails to address. To this end, we propose SocialEval, a script-based bilingual SI benchmark, integrating outcome- and process-oriented evaluation by manually c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.00900","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.00900/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.00900","created_at":"2026-07-05T11:13:42.543447+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.00900v1","created_at":"2026-07-05T11:13:42.543447+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.00900","created_at":"2026-07-05T11:13:42.543447+00:00"},{"alias_kind":"pith_short_12","alias_value":"FRBLHKS4WLS2","created_at":"2026-07-05T11:13:42.543447+00:00"},{"alias_kind":"pith_short_16","alias_value":"FRBLHKS4WLS2EBFC","created_at":"2026-07-05T11:13:42.543447+00:00"},{"alias_kind":"pith_short_8","alias_value":"FRBLHKS4","created_at":"2026-07-05T11:13:42.543447+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06157","citing_title":"LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability","ref_index":179,"is_internal_anchor":true},{"citing_arxiv_id":"2604.18982","citing_title":"SAVOIR: Learning Social Savoir-Faire via Shapley-based Reward Attribution","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FRBLHKS4WLS2EBFCFXQIWAOCZW","json":"https://pith.science/pith/FRBLHKS4WLS2EBFCFXQIWAOCZW.json","graph_json":"https://pith.science/api/pith-number/FRBLHKS4WLS2EBFCFXQIWAOCZW/graph.json","events_json":"https://pith.science/api/pith-number/FRBLHKS4WLS2EBFCFXQIWAOCZW/events.json","paper":"https://pith.science/paper/FRBLHKS4"},"agent_actions":{"view_html":"https://pith.science/pith/FRBLHKS4WLS2EBFCFXQIWAOCZW","download_json":"https://pith.science/pith/FRBLHKS4WLS2EBFCFXQIWAOCZW.json","view_paper":"https://pith.science/paper/FRBLHKS4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.00900&json=true","fetch_graph":"https://pith.science/api/pith-number/FRBLHKS4WLS2EBFCFXQIWAOCZW/graph.json","fetch_events":"https://pith.science/api/pith-number/FRBLHKS4WLS2EBFCFXQIWAOCZW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FRBLHKS4WLS2EBFCFXQIWAOCZW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FRBLHKS4WLS2EBFCFXQIWAOCZW/action/storage_attestation","attest_author":"https://pith.science/pith/FRBLHKS4WLS2EBFCFXQIWAOCZW/action/author_attestation","sign_citation":"https://pith.science/pith/FRBLHKS4WLS2EBFCFXQIWAOCZW/action/citation_signature","submit_replication":"https://pith.science/pith/FRBLHKS4WLS2EBFCFXQIWAOCZW/action/replication_record"}},"created_at":"2026-07-05T11:13:42.543447+00:00","updated_at":"2026-07-05T11:13:42.543447+00:00"}