{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:EVMTCGJXYWVDILLMXUFT4DJVE2","short_pith_number":"pith:EVMTCGJX","schema_version":"1.0","canonical_sha256":"2559311937c5aa342d6cbd0b3e0d3526b0f7fff75aaa6c7287021818b4207f4f","source":{"kind":"arxiv","id":"1912.11077","version":1},"attestation_state":"computed","paper":{"title":"Discrete and Continuous Action Representation for Practical RL in Video Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Adrien Logut, Eloi Alonso, Maxim Peter, Olivier Delalleau","submitted_at":"2019-12-23T19:37:13Z","abstract_excerpt":"While most current research in Reinforcement Learning (RL) focuses on improving the performance of the algorithms in controlled environments, the use of RL under constraints like those met in the video game industry is rarely studied. Operating under such constraints, we propose Hybrid SAC, an extension of the Soft Actor-Critic algorithm able to handle discrete, continuous and parameterized actions in a principled way. We show that Hybrid SAC can successfully solve a highspeed driving task in one of our games, and is competitive with the state-of-the-art on parameterized actions benchmark task"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.11077","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-12-23T19:37:13Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"cc8c6c994dfc622ac1a8a07466464c8b92d9df3ebf67d243a19d8d1720d6fd87","abstract_canon_sha256":"099953c8f005b3853557f0e31082eab6fee030a086f56e6e43e40448966c3309"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:28:16.751751Z","signature_b64":"4peFAnbhE9Vr3xPFVKBHNAmsxzIjSyuIjpt6zRyt4c3GNXR8KSjlgIDfXalImyq76eMVaQYxeCDRtIA5aBKNCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2559311937c5aa342d6cbd0b3e0d3526b0f7fff75aaa6c7287021818b4207f4f","last_reissued_at":"2026-07-05T00:28:16.751282Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:28:16.751282Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Discrete and Continuous Action Representation for Practical RL in Video Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Adrien Logut, Eloi Alonso, Maxim Peter, Olivier Delalleau","submitted_at":"2019-12-23T19:37:13Z","abstract_excerpt":"While most current research in Reinforcement Learning (RL) focuses on improving the performance of the algorithms in controlled environments, the use of RL under constraints like those met in the video game industry is rarely studied. Operating under such constraints, we propose Hybrid SAC, an extension of the Soft Actor-Critic algorithm able to handle discrete, continuous and parameterized actions in a principled way. We show that Hybrid SAC can successfully solve a highspeed driving task in one of our games, and is competitive with the state-of-the-art on parameterized actions benchmark task"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.11077","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.11077/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.11077","created_at":"2026-07-05T00:28:16.751361+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.11077v1","created_at":"2026-07-05T00:28:16.751361+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.11077","created_at":"2026-07-05T00:28:16.751361+00:00"},{"alias_kind":"pith_short_12","alias_value":"EVMTCGJXYWVD","created_at":"2026-07-05T00:28:16.751361+00:00"},{"alias_kind":"pith_short_16","alias_value":"EVMTCGJXYWVDILLM","created_at":"2026-07-05T00:28:16.751361+00:00"},{"alias_kind":"pith_short_8","alias_value":"EVMTCGJX","created_at":"2026-07-05T00:28:16.751361+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07498","citing_title":"Reward-Adaptive Iterative Discovery: A Case Study on Automated Game Testing for NHL26","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26574","citing_title":"Revisiting Action Factorization for Complex Action Spaces","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10601","citing_title":"Dmsh: A Multi-Agent Reinforcement Learning Framework for All-Quad Mesh Generation","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EVMTCGJXYWVDILLMXUFT4DJVE2","json":"https://pith.science/pith/EVMTCGJXYWVDILLMXUFT4DJVE2.json","graph_json":"https://pith.science/api/pith-number/EVMTCGJXYWVDILLMXUFT4DJVE2/graph.json","events_json":"https://pith.science/api/pith-number/EVMTCGJXYWVDILLMXUFT4DJVE2/events.json","paper":"https://pith.science/paper/EVMTCGJX"},"agent_actions":{"view_html":"https://pith.science/pith/EVMTCGJXYWVDILLMXUFT4DJVE2","download_json":"https://pith.science/pith/EVMTCGJXYWVDILLMXUFT4DJVE2.json","view_paper":"https://pith.science/paper/EVMTCGJX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.11077&json=true","fetch_graph":"https://pith.science/api/pith-number/EVMTCGJXYWVDILLMXUFT4DJVE2/graph.json","fetch_events":"https://pith.science/api/pith-number/EVMTCGJXYWVDILLMXUFT4DJVE2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EVMTCGJXYWVDILLMXUFT4DJVE2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EVMTCGJXYWVDILLMXUFT4DJVE2/action/storage_attestation","attest_author":"https://pith.science/pith/EVMTCGJXYWVDILLMXUFT4DJVE2/action/author_attestation","sign_citation":"https://pith.science/pith/EVMTCGJXYWVDILLMXUFT4DJVE2/action/citation_signature","submit_replication":"https://pith.science/pith/EVMTCGJXYWVDILLMXUFT4DJVE2/action/replication_record"}},"created_at":"2026-07-05T00:28:16.751361+00:00","updated_at":"2026-07-05T00:28:16.751361+00:00"}