{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:5EX5QDBLBBSAOGPWCWJJEG3BKF","short_pith_number":"pith:5EX5QDBL","schema_version":"1.0","canonical_sha256":"e92fd80c2b08640719f61592921b61514228f41ba0b6ece923c22afaf48bded1","source":{"kind":"arxiv","id":"1910.04376","version":2},"attestation_state":"computed","paper":{"title":"RLCard: A Toolkit for Reinforcement Learning in Card Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Daochen Zha, Junyu Guo, Kwei-Herng Lai, Ruzhe Wei, Songyi Huang, Xia Hu, Yuanpu Cao","submitted_at":"2019-10-10T05:56:16Z","abstract_excerpt":"RLCard is an open-source toolkit for reinforcement learning research in card games. It supports various card environments with easy-to-use interfaces, including Blackjack, Leduc Hold'em, Texas Hold'em, UNO, Dou Dizhu and Mahjong. The goal of RLCard is to bridge reinforcement learning and imperfect information games, and push forward the research of reinforcement learning in domains with multiple agents, large state and action space, and sparse reward. In this paper, we provide an overview of the key components in RLCard, a discussion of the design principles, a brief introduction of the interf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.04376","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2019-10-10T05:56:16Z","cross_cats_sorted":[],"title_canon_sha256":"04e44a2896224c0bcfa92b47c8335d0ed9b90adf89fb8b4756b9fe786d64edf0","abstract_canon_sha256":"4ff419cdadb62b6ae1cddd1dc9fdcbb8f24ac74998e33ede0940cbf1c7a2ec0e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:40:44.914807Z","signature_b64":"D4am3eClkTXoQA+okae//2qVNZrDJ8OgdTuAU7vMJoQx8OSWdbCL1XPWhpxebMPFx785MR4Lv2ZNg3W45cy2AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e92fd80c2b08640719f61592921b61514228f41ba0b6ece923c22afaf48bded1","last_reissued_at":"2026-07-05T00:40:44.914245Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:40:44.914245Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLCard: A Toolkit for Reinforcement Learning in Card Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Daochen Zha, Junyu Guo, Kwei-Herng Lai, Ruzhe Wei, Songyi Huang, Xia Hu, Yuanpu Cao","submitted_at":"2019-10-10T05:56:16Z","abstract_excerpt":"RLCard is an open-source toolkit for reinforcement learning research in card games. It supports various card environments with easy-to-use interfaces, including Blackjack, Leduc Hold'em, Texas Hold'em, UNO, Dou Dizhu and Mahjong. The goal of RLCard is to bridge reinforcement learning and imperfect information games, and push forward the research of reinforcement learning in domains with multiple agents, large state and action space, and sparse reward. In this paper, we provide an overview of the key components in RLCard, a discussion of the design principles, a brief introduction of the interf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.04376","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.04376/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.04376","created_at":"2026-07-05T00:40:44.914308+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.04376v2","created_at":"2026-07-05T00:40:44.914308+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.04376","created_at":"2026-07-05T00:40:44.914308+00:00"},{"alias_kind":"pith_short_12","alias_value":"5EX5QDBLBBSA","created_at":"2026-07-05T00:40:44.914308+00:00"},{"alias_kind":"pith_short_16","alias_value":"5EX5QDBLBBSAOGPW","created_at":"2026-07-05T00:40:44.914308+00:00"},{"alias_kind":"pith_short_8","alias_value":"5EX5QDBL","created_at":"2026-07-05T00:40:44.914308+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08656","citing_title":"From Player to Master: Enhancing Test-Time Learning of LLM Agents via Reinforcement Learning over Memory","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19235","citing_title":"GAE Falls Short in Imperfect-Information Self-Play Reinforcement Learning","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5EX5QDBLBBSAOGPWCWJJEG3BKF","json":"https://pith.science/pith/5EX5QDBLBBSAOGPWCWJJEG3BKF.json","graph_json":"https://pith.science/api/pith-number/5EX5QDBLBBSAOGPWCWJJEG3BKF/graph.json","events_json":"https://pith.science/api/pith-number/5EX5QDBLBBSAOGPWCWJJEG3BKF/events.json","paper":"https://pith.science/paper/5EX5QDBL"},"agent_actions":{"view_html":"https://pith.science/pith/5EX5QDBLBBSAOGPWCWJJEG3BKF","download_json":"https://pith.science/pith/5EX5QDBLBBSAOGPWCWJJEG3BKF.json","view_paper":"https://pith.science/paper/5EX5QDBL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.04376&json=true","fetch_graph":"https://pith.science/api/pith-number/5EX5QDBLBBSAOGPWCWJJEG3BKF/graph.json","fetch_events":"https://pith.science/api/pith-number/5EX5QDBLBBSAOGPWCWJJEG3BKF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5EX5QDBLBBSAOGPWCWJJEG3BKF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5EX5QDBLBBSAOGPWCWJJEG3BKF/action/storage_attestation","attest_author":"https://pith.science/pith/5EX5QDBLBBSAOGPWCWJJEG3BKF/action/author_attestation","sign_citation":"https://pith.science/pith/5EX5QDBLBBSAOGPWCWJJEG3BKF/action/citation_signature","submit_replication":"https://pith.science/pith/5EX5QDBLBBSAOGPWCWJJEG3BKF/action/replication_record"}},"created_at":"2026-07-05T00:40:44.914308+00:00","updated_at":"2026-07-05T00:40:44.914308+00:00"}