{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UDOAW2XRBQXI3RUAQ7BANCXFFL","short_pith_number":"pith:UDOAW2XR","schema_version":"1.0","canonical_sha256":"a0dc0b6af10c2e8dc68087c2068ae52acde4af4c8df91d27280abd377780ba4f","source":{"kind":"arxiv","id":"2304.02868","version":2},"attestation_state":"computed","paper":{"title":"Can Large Language Models Play Text Games Well? Current State-of-the-Art and Open Questions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chen Feng Tsai, Hongyuan Mei, Jing Li, Mo Yu, Sierra S. Liu, Xiaochen Zhou","submitted_at":"2023-04-06T05:01:28Z","abstract_excerpt":"Large language models (LLMs) such as ChatGPT and GPT-4 have recently demonstrated their remarkable abilities of communicating with human users. In this technical report, we take an initiative to investigate their capacities of playing text games, in which a player has to understand the environment and respond to situations by having dialogues with the game world. Our experiments show that ChatGPT performs competitively compared to all the existing systems but still exhibits a low level of intelligence. Precisely, ChatGPT can not construct the world model by playing the game or even reading the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.02868","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-04-06T05:01:28Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"6b0ef0d24be00c2dddcfd46536b9fec6d0551936f3c6f10aad22ba97d4d6deee","abstract_canon_sha256":"4a78baee29d07a0e6d3e352e12aaba6247f0071763476e382fd1e50f562da6a7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:09.625375Z","signature_b64":"guogASo8ZsNo3qMXVQLsCP+n7j2PuGRegzZo63gRHap82P7U+7srst67pMJGBXfxOEjlXVFKRpotBfUmy2brBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a0dc0b6af10c2e8dc68087c2068ae52acde4af4c8df91d27280abd377780ba4f","last_reissued_at":"2026-07-05T10:41:09.624874Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:09.624874Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Large Language Models Play Text Games Well? Current State-of-the-Art and Open Questions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chen Feng Tsai, Hongyuan Mei, Jing Li, Mo Yu, Sierra S. Liu, Xiaochen Zhou","submitted_at":"2023-04-06T05:01:28Z","abstract_excerpt":"Large language models (LLMs) such as ChatGPT and GPT-4 have recently demonstrated their remarkable abilities of communicating with human users. In this technical report, we take an initiative to investigate their capacities of playing text games, in which a player has to understand the environment and respond to situations by having dialogues with the game world. Our experiments show that ChatGPT performs competitively compared to all the existing systems but still exhibits a low level of intelligence. Precisely, ChatGPT can not construct the world model by playing the game or even reading the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.02868","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.02868/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.02868","created_at":"2026-07-05T10:41:09.624930+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.02868v2","created_at":"2026-07-05T10:41:09.624930+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.02868","created_at":"2026-07-05T10:41:09.624930+00:00"},{"alias_kind":"pith_short_12","alias_value":"UDOAW2XRBQXI","created_at":"2026-07-05T10:41:09.624930+00:00"},{"alias_kind":"pith_short_16","alias_value":"UDOAW2XRBQXI3RUA","created_at":"2026-07-05T10:41:09.624930+00:00"},{"alias_kind":"pith_short_8","alias_value":"UDOAW2XR","created_at":"2026-07-05T10:41:09.624930+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09826","citing_title":"OmniGameArena: A Unified UE5 Benchmark for VLM Game Agents with Improvement Dynamics","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2601.23206","citing_title":"High-quality generation of dynamic game content via small language models: A proof of concept","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2505.18134","citing_title":"VideoGameBench: Can Vision-Language Models complete popular video games?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2506.03610","citing_title":"Orak: A Foundational Benchmark for Training and Evaluating LLM Agents on Diverse Video Games","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UDOAW2XRBQXI3RUAQ7BANCXFFL","json":"https://pith.science/pith/UDOAW2XRBQXI3RUAQ7BANCXFFL.json","graph_json":"https://pith.science/api/pith-number/UDOAW2XRBQXI3RUAQ7BANCXFFL/graph.json","events_json":"https://pith.science/api/pith-number/UDOAW2XRBQXI3RUAQ7BANCXFFL/events.json","paper":"https://pith.science/paper/UDOAW2XR"},"agent_actions":{"view_html":"https://pith.science/pith/UDOAW2XRBQXI3RUAQ7BANCXFFL","download_json":"https://pith.science/pith/UDOAW2XRBQXI3RUAQ7BANCXFFL.json","view_paper":"https://pith.science/paper/UDOAW2XR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.02868&json=true","fetch_graph":"https://pith.science/api/pith-number/UDOAW2XRBQXI3RUAQ7BANCXFFL/graph.json","fetch_events":"https://pith.science/api/pith-number/UDOAW2XRBQXI3RUAQ7BANCXFFL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UDOAW2XRBQXI3RUAQ7BANCXFFL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UDOAW2XRBQXI3RUAQ7BANCXFFL/action/storage_attestation","attest_author":"https://pith.science/pith/UDOAW2XRBQXI3RUAQ7BANCXFFL/action/author_attestation","sign_citation":"https://pith.science/pith/UDOAW2XRBQXI3RUAQ7BANCXFFL/action/citation_signature","submit_replication":"https://pith.science/pith/UDOAW2XRBQXI3RUAQ7BANCXFFL/action/replication_record"}},"created_at":"2026-07-05T10:41:09.624930+00:00","updated_at":"2026-07-05T10:41:09.624930+00:00"}