{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RBDWO4FIRYMXFJSVLBG2PE3VFY","short_pith_number":"pith:RBDWO4FI","schema_version":"1.0","canonical_sha256":"88476770a88e1972a655584da793752e37cf5cd5810e0f89e5a4059eaf3616f8","source":{"kind":"arxiv","id":"2501.13883","version":2},"attestation_state":"computed","paper":{"title":"Utilizing Evolution Strategies to Train Transformers in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.LG","authors_text":"Maty\\'a\\v{s} Lorenc, Roman Neruda","submitted_at":"2025-01-23T17:56:40Z","abstract_excerpt":"We explore the capability of evolution strategies to train an agent with a policy based on a transformer architecture in a reinforcement learning setting. We performed experiments using OpenAI's highly parallelizable evolution strategy to train Decision Transformer in the MuJoCo Humanoid locomotion environment and in the environment of Atari games, testing the ability of this black-box optimization technique to train even such relatively large and complicated models (compared to those previously tested in the literature). The examined evolution strategy proved to be, in general, capable of ach"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.13883","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-01-23T17:56:40Z","cross_cats_sorted":["cs.NE"],"title_canon_sha256":"eb91ea76e07a17fc725547cd0a21658245fa50023545ebf60a14a97b0535abe7","abstract_canon_sha256":"0c9920126663d74a04695514cbe6abf77b2cbeccdb6d4e98fdc615dc9ae4fe7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:30.998377Z","signature_b64":"IPIS/h09C/NpgFEI0jH1oAMxyZsD7mv3bdh+ioGE5ZTQKKXProU6kZHYEBAvYUc2oxFz4bzidCr2B69kDt2kCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"88476770a88e1972a655584da793752e37cf5cd5810e0f89e5a4059eaf3616f8","last_reissued_at":"2026-07-05T11:45:30.997848Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:30.997848Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Utilizing Evolution Strategies to Train Transformers in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.LG","authors_text":"Maty\\'a\\v{s} Lorenc, Roman Neruda","submitted_at":"2025-01-23T17:56:40Z","abstract_excerpt":"We explore the capability of evolution strategies to train an agent with a policy based on a transformer architecture in a reinforcement learning setting. We performed experiments using OpenAI's highly parallelizable evolution strategy to train Decision Transformer in the MuJoCo Humanoid locomotion environment and in the environment of Atari games, testing the ability of this black-box optimization technique to train even such relatively large and complicated models (compared to those previously tested in the literature). The examined evolution strategy proved to be, in general, capable of ach"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.13883","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.13883/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.13883","created_at":"2026-07-05T11:45:30.997900+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.13883v2","created_at":"2026-07-05T11:45:30.997900+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.13883","created_at":"2026-07-05T11:45:30.997900+00:00"},{"alias_kind":"pith_short_12","alias_value":"RBDWO4FIRYMX","created_at":"2026-07-05T11:45:30.997900+00:00"},{"alias_kind":"pith_short_16","alias_value":"RBDWO4FIRYMXFJSV","created_at":"2026-07-05T11:45:30.997900+00:00"},{"alias_kind":"pith_short_8","alias_value":"RBDWO4FI","created_at":"2026-07-05T11:45:30.997900+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RBDWO4FIRYMXFJSVLBG2PE3VFY","json":"https://pith.science/pith/RBDWO4FIRYMXFJSVLBG2PE3VFY.json","graph_json":"https://pith.science/api/pith-number/RBDWO4FIRYMXFJSVLBG2PE3VFY/graph.json","events_json":"https://pith.science/api/pith-number/RBDWO4FIRYMXFJSVLBG2PE3VFY/events.json","paper":"https://pith.science/paper/RBDWO4FI"},"agent_actions":{"view_html":"https://pith.science/pith/RBDWO4FIRYMXFJSVLBG2PE3VFY","download_json":"https://pith.science/pith/RBDWO4FIRYMXFJSVLBG2PE3VFY.json","view_paper":"https://pith.science/paper/RBDWO4FI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.13883&json=true","fetch_graph":"https://pith.science/api/pith-number/RBDWO4FIRYMXFJSVLBG2PE3VFY/graph.json","fetch_events":"https://pith.science/api/pith-number/RBDWO4FIRYMXFJSVLBG2PE3VFY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RBDWO4FIRYMXFJSVLBG2PE3VFY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RBDWO4FIRYMXFJSVLBG2PE3VFY/action/storage_attestation","attest_author":"https://pith.science/pith/RBDWO4FIRYMXFJSVLBG2PE3VFY/action/author_attestation","sign_citation":"https://pith.science/pith/RBDWO4FIRYMXFJSVLBG2PE3VFY/action/citation_signature","submit_replication":"https://pith.science/pith/RBDWO4FIRYMXFJSVLBG2PE3VFY/action/replication_record"}},"created_at":"2026-07-05T11:45:30.997900+00:00","updated_at":"2026-07-05T11:45:30.997900+00:00"}