{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KNNMMJA3QS2RAP3GTSEGSIVCEA","short_pith_number":"pith:KNNMMJA3","schema_version":"1.0","canonical_sha256":"535ac6241b84b5103f669c886922a22010256a67634861b52fee5094137e4103","source":{"kind":"arxiv","id":"2306.09200","version":2},"attestation_state":"computed","paper":{"title":"ChessGPT: Bridging Policy Learning and Language Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"David Mguni, Hongrui Tang, Jun Wang, Kun Shao, Mengyue Yang, Xidong Feng, Yali Du, Yicheng Luo, Ziyan Wang","submitted_at":"2023-06-15T15:35:31Z","abstract_excerpt":"When solving decision-making tasks, humans typically depend on information from two key sources: (1) Historical policy data, which provides interaction replay from the environment, and (2) Analytical insights in natural language form, exposing the invaluable thought process or strategic considerations. Despite this, the majority of preceding research focuses on only one source: they either use historical replay exclusively to directly learn policy or value functions, or engaged in language model training utilizing mere language corpus. In this paper, we argue that a powerful autonomous agent s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.09200","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-15T15:35:31Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ec2827f55f599e738f876d9e31f5d40498e77d532e08df66f41bd5c644f2c011","abstract_canon_sha256":"91f42372fb8ea5bb2d93a6b7d3078b91431edb99fcd1b257e7d61365c2b11b47"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:26:55.479397Z","signature_b64":"5i5uBgyk58RORUJNNm5BvO/CyzjFMCGCG7/k3JNGCbaFm2tIHrdJzahgXaotH9yB6GGXeqCTz+FTE5reQUgWBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"535ac6241b84b5103f669c886922a22010256a67634861b52fee5094137e4103","last_reissued_at":"2026-07-05T07:26:55.478846Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:26:55.478846Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChessGPT: Bridging Policy Learning and Language Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"David Mguni, Hongrui Tang, Jun Wang, Kun Shao, Mengyue Yang, Xidong Feng, Yali Du, Yicheng Luo, Ziyan Wang","submitted_at":"2023-06-15T15:35:31Z","abstract_excerpt":"When solving decision-making tasks, humans typically depend on information from two key sources: (1) Historical policy data, which provides interaction replay from the environment, and (2) Analytical insights in natural language form, exposing the invaluable thought process or strategic considerations. Despite this, the majority of preceding research focuses on only one source: they either use historical replay exclusively to directly learn policy or value functions, or engaged in language model training utilizing mere language corpus. In this paper, we argue that a powerful autonomous agent s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.09200","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.09200/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.09200","created_at":"2026-07-05T07:26:55.478905+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.09200v2","created_at":"2026-07-05T07:26:55.478905+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.09200","created_at":"2026-07-05T07:26:55.478905+00:00"},{"alias_kind":"pith_short_12","alias_value":"KNNMMJA3QS2R","created_at":"2026-07-05T07:26:55.478905+00:00"},{"alias_kind":"pith_short_16","alias_value":"KNNMMJA3QS2RAP3G","created_at":"2026-07-05T07:26:55.478905+00:00"},{"alias_kind":"pith_short_8","alias_value":"KNNMMJA3","created_at":"2026-07-05T07:26:55.478905+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23668","citing_title":"On the Limits of Prompt-Conditioned Language Models as General-Purpose Learners","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24375","citing_title":"Distilling Game Code World Model Generation into Lightweight Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2511.03724","citing_title":"Outbidding and Outbluffing Elite Humans: Mastering Liar's Poker via Self-Play and Reinforcement Learning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07752","citing_title":"MIMIC-Py: An Extensible Tool for Personality-Driven Automated Game Testing with Large Language Models","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KNNMMJA3QS2RAP3GTSEGSIVCEA","json":"https://pith.science/pith/KNNMMJA3QS2RAP3GTSEGSIVCEA.json","graph_json":"https://pith.science/api/pith-number/KNNMMJA3QS2RAP3GTSEGSIVCEA/graph.json","events_json":"https://pith.science/api/pith-number/KNNMMJA3QS2RAP3GTSEGSIVCEA/events.json","paper":"https://pith.science/paper/KNNMMJA3"},"agent_actions":{"view_html":"https://pith.science/pith/KNNMMJA3QS2RAP3GTSEGSIVCEA","download_json":"https://pith.science/pith/KNNMMJA3QS2RAP3GTSEGSIVCEA.json","view_paper":"https://pith.science/paper/KNNMMJA3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.09200&json=true","fetch_graph":"https://pith.science/api/pith-number/KNNMMJA3QS2RAP3GTSEGSIVCEA/graph.json","fetch_events":"https://pith.science/api/pith-number/KNNMMJA3QS2RAP3GTSEGSIVCEA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KNNMMJA3QS2RAP3GTSEGSIVCEA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KNNMMJA3QS2RAP3GTSEGSIVCEA/action/storage_attestation","attest_author":"https://pith.science/pith/KNNMMJA3QS2RAP3GTSEGSIVCEA/action/author_attestation","sign_citation":"https://pith.science/pith/KNNMMJA3QS2RAP3GTSEGSIVCEA/action/citation_signature","submit_replication":"https://pith.science/pith/KNNMMJA3QS2RAP3GTSEGSIVCEA/action/replication_record"}},"created_at":"2026-07-05T07:26:55.478905+00:00","updated_at":"2026-07-05T07:26:55.478905+00:00"}