{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VN5JC66W5XA2F3EBWA33NV7H42","short_pith_number":"pith:VN5JC66W","schema_version":"1.0","canonical_sha256":"ab7a917bd6edc1a2ec81b037b6d7e7e6944f867b4b5050feaecdac1f29afa15e","source":{"kind":"arxiv","id":"2502.00595","version":1},"attestation_state":"computed","paper":{"title":"RPGBENCH: Evaluating Large Language Models as Role-Playing Game Engines","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alex Smola, Andrea Yaoyun Cui, Dongming Shen, Jaewon Lee, Mu Li, Pengfei Yu, Silin Meng, Weisu Yin, Xingjian Shi, Yi Zhu, Zhenlin Xu","submitted_at":"2025-02-01T23:40:24Z","abstract_excerpt":"We present RPGBench, the first benchmark designed to evaluate large language models (LLMs) as text-based role-playing game (RPG) engines. RPGBench comprises two core tasks: Game Creation (GC) and Game Simulation (GS). In GC, an LLM must craft a valid and playable RPG world using a structured event-state representation, ensuring logical coherence and proper termination conditions. In GS, the LLM simulates interactive gameplay across multiple rounds while consistently updating states and enforcing game rules. To comprehensively assess performance, RPGBench integrates objective and subjective eva"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.00595","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-01T23:40:24Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ff1a5a3032b45e8bb4f9d5ff9355a7f3d5ba53a8a31e14a5bdc987a345b75769","abstract_canon_sha256":"773823fa9fdbe095f050343cb02b2653c4f91370cafd5b23521fa93effc33858"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:35.906240Z","signature_b64":"12q+oy0c77LgaDFOUVTTWzQ1CcOeN372Y+Y6QulyUSb6JP79/fPY9scoFbzz/yDeyq/l1zjZlHMQs3lardAVBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab7a917bd6edc1a2ec81b037b6d7e7e6944f867b4b5050feaecdac1f29afa15e","last_reissued_at":"2026-07-05T10:08:35.905821Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:35.905821Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RPGBENCH: Evaluating Large Language Models as Role-Playing Game Engines","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alex Smola, Andrea Yaoyun Cui, Dongming Shen, Jaewon Lee, Mu Li, Pengfei Yu, Silin Meng, Weisu Yin, Xingjian Shi, Yi Zhu, Zhenlin Xu","submitted_at":"2025-02-01T23:40:24Z","abstract_excerpt":"We present RPGBench, the first benchmark designed to evaluate large language models (LLMs) as text-based role-playing game (RPG) engines. RPGBench comprises two core tasks: Game Creation (GC) and Game Simulation (GS). In GC, an LLM must craft a valid and playable RPG world using a structured event-state representation, ensuring logical coherence and proper termination conditions. In GS, the LLM simulates interactive gameplay across multiple rounds while consistently updating states and enforcing game rules. To comprehensively assess performance, RPGBench integrates objective and subjective eva"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00595","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.00595/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.00595","created_at":"2026-07-05T10:08:35.905879+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.00595v1","created_at":"2026-07-05T10:08:35.905879+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00595","created_at":"2026-07-05T10:08:35.905879+00:00"},{"alias_kind":"pith_short_12","alias_value":"VN5JC66W5XA2","created_at":"2026-07-05T10:08:35.905879+00:00"},{"alias_kind":"pith_short_16","alias_value":"VN5JC66W5XA2F3EB","created_at":"2026-07-05T10:08:35.905879+00:00"},{"alias_kind":"pith_short_8","alias_value":"VN5JC66W","created_at":"2026-07-05T10:08:35.905879+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.08565","citing_title":"Moral Susceptibility and Robustness under Persona Role-Play in Large Language Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14169","citing_title":"BOOKMARKS: Efficient Active Storyline Memory for Role-playing","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08340","citing_title":"Mastering PokeGym: Graph-Guided Multimodal Evolution at Test Time","ref_index":87,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VN5JC66W5XA2F3EBWA33NV7H42","json":"https://pith.science/pith/VN5JC66W5XA2F3EBWA33NV7H42.json","graph_json":"https://pith.science/api/pith-number/VN5JC66W5XA2F3EBWA33NV7H42/graph.json","events_json":"https://pith.science/api/pith-number/VN5JC66W5XA2F3EBWA33NV7H42/events.json","paper":"https://pith.science/paper/VN5JC66W"},"agent_actions":{"view_html":"https://pith.science/pith/VN5JC66W5XA2F3EBWA33NV7H42","download_json":"https://pith.science/pith/VN5JC66W5XA2F3EBWA33NV7H42.json","view_paper":"https://pith.science/paper/VN5JC66W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.00595&json=true","fetch_graph":"https://pith.science/api/pith-number/VN5JC66W5XA2F3EBWA33NV7H42/graph.json","fetch_events":"https://pith.science/api/pith-number/VN5JC66W5XA2F3EBWA33NV7H42/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VN5JC66W5XA2F3EBWA33NV7H42/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VN5JC66W5XA2F3EBWA33NV7H42/action/storage_attestation","attest_author":"https://pith.science/pith/VN5JC66W5XA2F3EBWA33NV7H42/action/author_attestation","sign_citation":"https://pith.science/pith/VN5JC66W5XA2F3EBWA33NV7H42/action/citation_signature","submit_replication":"https://pith.science/pith/VN5JC66W5XA2F3EBWA33NV7H42/action/replication_record"}},"created_at":"2026-07-05T10:08:35.905879+00:00","updated_at":"2026-07-05T10:08:35.905879+00:00"}