{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EEKY3OVEYZD43GLEPQJ6BW5GEQ","short_pith_number":"pith:EEKY3OVE","schema_version":"1.0","canonical_sha256":"21158dbaa4c647cd99647c13e0dba6241f37132b5ff3069dd445ded6e62b1172","source":{"kind":"arxiv","id":"2412.13602","version":2},"attestation_state":"computed","paper":{"title":"GAMEBoT: Transparent Assessment of LLM Reasoning in Games","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jonathan Roberts, Kai Han, Samuel Albanie, Wenye Lin, Yunhan Yang, Zongqing Lu","submitted_at":"2024-12-18T08:32:53Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly deployed in real-world applications that demand complex reasoning. To track progress, robust benchmarks are required to evaluate their capabilities beyond superficial pattern recognition. However, current LLM reasoning benchmarks often face challenges such as insufficient interpretability, performance saturation or data contamination. To address these challenges, we introduce GAMEBoT, a gaming arena designed for rigorous and transparent assessment of LLM reasoning capabilities. GAMEBoT decomposes complex reasoning in games into predefined modular s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.13602","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-18T08:32:53Z","cross_cats_sorted":[],"title_canon_sha256":"4f44fbe5aa3fa186b914eefb7217c6446c222d892aeecd14396c6cf1fb046eca","abstract_canon_sha256":"ef995dcd94ce21a8bfed7ebaf43f25743be5c4af2e14bfd3967b8bfe56bda377"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:17.257166Z","signature_b64":"N15XPunX2tzmRe8lLKKoUNwwJ6O596ViX6M7us295VU2iW/wb0iB3FgQr4jILfbn/joRQaM+0R23HMUNJKg0Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"21158dbaa4c647cd99647c13e0dba6241f37132b5ff3069dd445ded6e62b1172","last_reissued_at":"2026-07-05T11:13:17.256716Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:17.256716Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GAMEBoT: Transparent Assessment of LLM Reasoning in Games","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jonathan Roberts, Kai Han, Samuel Albanie, Wenye Lin, Yunhan Yang, Zongqing Lu","submitted_at":"2024-12-18T08:32:53Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly deployed in real-world applications that demand complex reasoning. To track progress, robust benchmarks are required to evaluate their capabilities beyond superficial pattern recognition. However, current LLM reasoning benchmarks often face challenges such as insufficient interpretability, performance saturation or data contamination. To address these challenges, we introduce GAMEBoT, a gaming arena designed for rigorous and transparent assessment of LLM reasoning capabilities. GAMEBoT decomposes complex reasoning in games into predefined modular s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.13602","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.13602/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.13602","created_at":"2026-07-05T11:13:17.256769+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.13602v2","created_at":"2026-07-05T11:13:17.256769+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.13602","created_at":"2026-07-05T11:13:17.256769+00:00"},{"alias_kind":"pith_short_12","alias_value":"EEKY3OVEYZD4","created_at":"2026-07-05T11:13:17.256769+00:00"},{"alias_kind":"pith_short_16","alias_value":"EEKY3OVEYZD43GLE","created_at":"2026-07-05T11:13:17.256769+00:00"},{"alias_kind":"pith_short_8","alias_value":"EEKY3OVE","created_at":"2026-07-05T11:13:17.256769+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12191","citing_title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2508.08636","citing_title":"InternBootcamp Technical Report: Boosting LLM Reasoning with Verifiable Task Scaling","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06079","citing_title":"Scientific Graphics Program Synthesis via Dual Self-Consistency Reinforcement Learning","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EEKY3OVEYZD43GLEPQJ6BW5GEQ","json":"https://pith.science/pith/EEKY3OVEYZD43GLEPQJ6BW5GEQ.json","graph_json":"https://pith.science/api/pith-number/EEKY3OVEYZD43GLEPQJ6BW5GEQ/graph.json","events_json":"https://pith.science/api/pith-number/EEKY3OVEYZD43GLEPQJ6BW5GEQ/events.json","paper":"https://pith.science/paper/EEKY3OVE"},"agent_actions":{"view_html":"https://pith.science/pith/EEKY3OVEYZD43GLEPQJ6BW5GEQ","download_json":"https://pith.science/pith/EEKY3OVEYZD43GLEPQJ6BW5GEQ.json","view_paper":"https://pith.science/paper/EEKY3OVE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.13602&json=true","fetch_graph":"https://pith.science/api/pith-number/EEKY3OVEYZD43GLEPQJ6BW5GEQ/graph.json","fetch_events":"https://pith.science/api/pith-number/EEKY3OVEYZD43GLEPQJ6BW5GEQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EEKY3OVEYZD43GLEPQJ6BW5GEQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EEKY3OVEYZD43GLEPQJ6BW5GEQ/action/storage_attestation","attest_author":"https://pith.science/pith/EEKY3OVEYZD43GLEPQJ6BW5GEQ/action/author_attestation","sign_citation":"https://pith.science/pith/EEKY3OVEYZD43GLEPQJ6BW5GEQ/action/citation_signature","submit_replication":"https://pith.science/pith/EEKY3OVEYZD43GLEPQJ6BW5GEQ/action/replication_record"}},"created_at":"2026-07-05T11:13:17.256769+00:00","updated_at":"2026-07-05T11:13:17.256769+00:00"}