{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:GKJKKCSSAT7GKLOE6T476IZUFD","short_pith_number":"pith:GKJKKCSS","schema_version":"1.0","canonical_sha256":"3292a50a5204fe652dc4f4f9ff233428d38ad90ca4858b96a63964043093b165","source":{"kind":"arxiv","id":"1912.03918","version":1},"attestation_state":"computed","paper":{"title":"Transformer Based Reinforcement Learning For Games","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.LG","authors_text":"Mayanka Medhe, Nikunj Shah, Sucheta Ravikanti, Uddeshya Upadhyay","submitted_at":"2019-12-09T09:35:48Z","abstract_excerpt":"Recent times have witnessed sharp improvements in reinforcement learning tasks using deep reinforcement learning techniques like Deep Q Networks, Policy Gradients, Actor Critic methods which are based on deep learning based models and back-propagation of gradients to train such models. An active area of research in reinforcement learning is about training agents to play complex video games, which so far has been something accomplished only by human intelligence. Some state of the art performances in video game playing using deep reinforcement learning are obtained by processing the sequence of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.03918","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2019-12-09T09:35:48Z","cross_cats_sorted":["cs.NE"],"title_canon_sha256":"25e77ab50379a08d6444383870b30c43de53b3f15fea609379ca47296b40b0d3","abstract_canon_sha256":"2b820e8bc196df777fd07b3381318b66e4f0bdd6d791d5325eee891c4ce95c3d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:24:50.025844Z","signature_b64":"uTzo6pynZ5j9VmMF2jfgt/BSXmOwFVi10gkbmOp0p4Pb1U9UyPPwxOTRqj0yt1SnIrNLkrQkM70bJ/Wx0e8qAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3292a50a5204fe652dc4f4f9ff233428d38ad90ca4858b96a63964043093b165","last_reissued_at":"2026-07-05T00:24:50.025481Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:24:50.025481Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformer Based Reinforcement Learning For Games","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.LG","authors_text":"Mayanka Medhe, Nikunj Shah, Sucheta Ravikanti, Uddeshya Upadhyay","submitted_at":"2019-12-09T09:35:48Z","abstract_excerpt":"Recent times have witnessed sharp improvements in reinforcement learning tasks using deep reinforcement learning techniques like Deep Q Networks, Policy Gradients, Actor Critic methods which are based on deep learning based models and back-propagation of gradients to train such models. An active area of research in reinforcement learning is about training agents to play complex video games, which so far has been something accomplished only by human intelligence. Some state of the art performances in video game playing using deep reinforcement learning are obtained by processing the sequence of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.03918","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.03918/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.03918","created_at":"2026-07-05T00:24:50.025536+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.03918v1","created_at":"2026-07-05T00:24:50.025536+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.03918","created_at":"2026-07-05T00:24:50.025536+00:00"},{"alias_kind":"pith_short_12","alias_value":"GKJKKCSSAT7G","created_at":"2026-07-05T00:24:50.025536+00:00"},{"alias_kind":"pith_short_16","alias_value":"GKJKKCSSAT7GKLOE","created_at":"2026-07-05T00:24:50.025536+00:00"},{"alias_kind":"pith_short_8","alias_value":"GKJKKCSS","created_at":"2026-07-05T00:24:50.025536+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.21956","citing_title":"Optimal Return-to-Go Guided Decision Transformer for Auto-Bidding in Advertisement","ref_index":2019,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GKJKKCSSAT7GKLOE6T476IZUFD","json":"https://pith.science/pith/GKJKKCSSAT7GKLOE6T476IZUFD.json","graph_json":"https://pith.science/api/pith-number/GKJKKCSSAT7GKLOE6T476IZUFD/graph.json","events_json":"https://pith.science/api/pith-number/GKJKKCSSAT7GKLOE6T476IZUFD/events.json","paper":"https://pith.science/paper/GKJKKCSS"},"agent_actions":{"view_html":"https://pith.science/pith/GKJKKCSSAT7GKLOE6T476IZUFD","download_json":"https://pith.science/pith/GKJKKCSSAT7GKLOE6T476IZUFD.json","view_paper":"https://pith.science/paper/GKJKKCSS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.03918&json=true","fetch_graph":"https://pith.science/api/pith-number/GKJKKCSSAT7GKLOE6T476IZUFD/graph.json","fetch_events":"https://pith.science/api/pith-number/GKJKKCSSAT7GKLOE6T476IZUFD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GKJKKCSSAT7GKLOE6T476IZUFD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GKJKKCSSAT7GKLOE6T476IZUFD/action/storage_attestation","attest_author":"https://pith.science/pith/GKJKKCSSAT7GKLOE6T476IZUFD/action/author_attestation","sign_citation":"https://pith.science/pith/GKJKKCSSAT7GKLOE6T476IZUFD/action/citation_signature","submit_replication":"https://pith.science/pith/GKJKKCSSAT7GKLOE6T476IZUFD/action/replication_record"}},"created_at":"2026-07-05T00:24:50.025536+00:00","updated_at":"2026-07-05T00:24:50.025536+00:00"}