{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YDWFOC5S47DFPZXUD7X2XP2ESF","short_pith_number":"pith:YDWFOC5S","schema_version":"1.0","canonical_sha256":"c0ec570bb2e7c657e6f41fefabbf4491469be1a450b3bf210588fe6d7428dc3f","source":{"kind":"arxiv","id":"2308.10032","version":1},"attestation_state":"computed","paper":{"title":"GameEval: Evaluating LLMs on Conversational Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenfei Wu, Dan Qiao, Juntao Li, Nan Duan, Yaobo Liang","submitted_at":"2023-08-19T14:33:40Z","abstract_excerpt":"The rapid advancements in large language models (LLMs) have presented challenges in evaluating those models. Existing evaluation methods are either reference-based or preference based, which inevitably need human intervention or introduce test bias caused by evaluator models. In this paper, we propose GameEval, a novel approach to evaluating LLMs through goal-driven conversational games, overcoming the limitations of previous methods. GameEval treats LLMs as game players and assigns them distinct roles with specific goals achieved by launching conversations of various forms, including discussi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.10032","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-19T14:33:40Z","cross_cats_sorted":[],"title_canon_sha256":"91b51c1b0fc97c8347f15b29b9359b82267ef7df3108123f52593bae4fec7da7","abstract_canon_sha256":"037c87bd1ee5ac9013ea3cd247164dd1f487337c69d9cc88e53ff9d51859fea6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:42:50.346015Z","signature_b64":"xp3lp8Em3amL+pjAJjoGyJcuL/+40Avd38Pg+5fKGFCyIqTLNXpLDJhdS48PoJGFnR46bSz+20XzPVNVLHjiDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0ec570bb2e7c657e6f41fefabbf4491469be1a450b3bf210588fe6d7428dc3f","last_reissued_at":"2026-07-05T06:42:50.345607Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:42:50.345607Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GameEval: Evaluating LLMs on Conversational Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenfei Wu, Dan Qiao, Juntao Li, Nan Duan, Yaobo Liang","submitted_at":"2023-08-19T14:33:40Z","abstract_excerpt":"The rapid advancements in large language models (LLMs) have presented challenges in evaluating those models. Existing evaluation methods are either reference-based or preference based, which inevitably need human intervention or introduce test bias caused by evaluator models. In this paper, we propose GameEval, a novel approach to evaluating LLMs through goal-driven conversational games, overcoming the limitations of previous methods. GameEval treats LLMs as game players and assigns them distinct roles with specific goals achieved by launching conversations of various forms, including discussi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.10032","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.10032/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.10032","created_at":"2026-07-05T06:42:50.345669+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.10032v1","created_at":"2026-07-05T06:42:50.345669+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.10032","created_at":"2026-07-05T06:42:50.345669+00:00"},{"alias_kind":"pith_short_12","alias_value":"YDWFOC5S47DF","created_at":"2026-07-05T06:42:50.345669+00:00"},{"alias_kind":"pith_short_16","alias_value":"YDWFOC5S47DFPZXU","created_at":"2026-07-05T06:42:50.345669+00:00"},{"alias_kind":"pith_short_8","alias_value":"YDWFOC5S","created_at":"2026-07-05T06:42:50.345669+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31387","citing_title":"Multi-Turn Multi-Agent Dialogue for Collaborative Reconstruction Improves VLM Performance on Spatial Reasoning, But Only Barely","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13875","citing_title":"Common-agency Games for Multi-Objective Test-Time Alignment","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YDWFOC5S47DFPZXUD7X2XP2ESF","json":"https://pith.science/pith/YDWFOC5S47DFPZXUD7X2XP2ESF.json","graph_json":"https://pith.science/api/pith-number/YDWFOC5S47DFPZXUD7X2XP2ESF/graph.json","events_json":"https://pith.science/api/pith-number/YDWFOC5S47DFPZXUD7X2XP2ESF/events.json","paper":"https://pith.science/paper/YDWFOC5S"},"agent_actions":{"view_html":"https://pith.science/pith/YDWFOC5S47DFPZXUD7X2XP2ESF","download_json":"https://pith.science/pith/YDWFOC5S47DFPZXUD7X2XP2ESF.json","view_paper":"https://pith.science/paper/YDWFOC5S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.10032&json=true","fetch_graph":"https://pith.science/api/pith-number/YDWFOC5S47DFPZXUD7X2XP2ESF/graph.json","fetch_events":"https://pith.science/api/pith-number/YDWFOC5S47DFPZXUD7X2XP2ESF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YDWFOC5S47DFPZXUD7X2XP2ESF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YDWFOC5S47DFPZXUD7X2XP2ESF/action/storage_attestation","attest_author":"https://pith.science/pith/YDWFOC5S47DFPZXUD7X2XP2ESF/action/author_attestation","sign_citation":"https://pith.science/pith/YDWFOC5S47DFPZXUD7X2XP2ESF/action/citation_signature","submit_replication":"https://pith.science/pith/YDWFOC5S47DFPZXUD7X2XP2ESF/action/replication_record"}},"created_at":"2026-07-05T06:42:50.345669+00:00","updated_at":"2026-07-05T06:42:50.345669+00:00"}