{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2018:6YCQBQHDQSECVYHR6JURETCC7L","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4bc6fb36deb8a324d3f9a433d42dd5dcaaf3a80f97c81baa52f9f3f15d1d394e","cross_cats_sorted":[],"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.AI","submitted_at":"2018-09-15T20:01:06Z","title_canon_sha256":"1179777ddf8bd7c92d29bd75834738dafe21aff1036fb866c1bf67a7a66c8315"},"schema_version":"1.0","source":{"id":"1809.05763","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1809.05763","created_at":"2026-05-18T00:05:36Z"},{"alias_kind":"arxiv_version","alias_value":"1809.05763v1","created_at":"2026-05-18T00:05:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1809.05763","created_at":"2026-05-18T00:05:36Z"},{"alias_kind":"pith_short_12","alias_value":"6YCQBQHDQSEC","created_at":"2026-05-18T12:32:11Z"},{"alias_kind":"pith_short_16","alias_value":"6YCQBQHDQSECVYHR","created_at":"2026-05-18T12:32:11Z"},{"alias_kind":"pith_short_8","alias_value":"6YCQBQHD","created_at":"2026-05-18T12:32:11Z"}],"graph_snapshots":[{"event_id":"sha256:1f185cf6cb38f74f6fa1b1c9e3d19d8a159f6538139b7bf95e58ecd57ff9b499","target":"graph","created_at":"2026-05-18T00:05:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"In this paper, a new offline actor-critic learning algorithm is introduced: Sampled Policy Gradient (SPG). SPG samples in the action space to calculate an approximated policy gradient by using the critic to evaluate the samples. This sampling allows SPG to search the action-Q-value space more globally than deterministic policy gradient (DPG), enabling it to theoretically avoid more local optima. SPG is compared to Q-learning and the actor-critic algorithms CACLA and DPG in a pellet collection task and a self play environment in the game Agar.io. The online game Agar.io has become massively pop","authors_text":"Anton Orell Wiehe, Madalina M. Drugan, Marco A. Wiering, Nil Stolt Ans\\'o","cross_cats":[],"headline":"","license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.AI","submitted_at":"2018-09-15T20:01:06Z","title":"Sampled Policy Gradient for Learning to Play the Game Agar.io"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1809.05763","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b6a310a94e0fd25f0b537f975b9598e68473835a9a3d6005aeff48aeda4d4928","target":"record","created_at":"2026-05-18T00:05:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4bc6fb36deb8a324d3f9a433d42dd5dcaaf3a80f97c81baa52f9f3f15d1d394e","cross_cats_sorted":[],"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.AI","submitted_at":"2018-09-15T20:01:06Z","title_canon_sha256":"1179777ddf8bd7c92d29bd75834738dafe21aff1036fb866c1bf67a7a66c8315"},"schema_version":"1.0","source":{"id":"1809.05763","kind":"arxiv","version":1}},"canonical_sha256":"f60500c0e384882ae0f1f269124c42fade74402fe583e32447960eccabfaa4e7","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f60500c0e384882ae0f1f269124c42fade74402fe583e32447960eccabfaa4e7","first_computed_at":"2026-05-18T00:05:36.583354Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T00:05:36.583354Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ViYMNTdxfAwnV51/9llwrzGAuwz+7+CoRs5gOgZ2330pDobjKqy15i8y2gdZy3NSYqACtvlrMz0nHjhmrWwoDA==","signature_status":"signed_v1","signed_at":"2026-05-18T00:05:36.583754Z","signed_message":"canonical_sha256_bytes"},"source_id":"1809.05763","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b6a310a94e0fd25f0b537f975b9598e68473835a9a3d6005aeff48aeda4d4928","sha256:1f185cf6cb38f74f6fa1b1c9e3d19d8a159f6538139b7bf95e58ecd57ff9b499"],"state_sha256":"0a2ecc68e03636399079120f7a9d509a6c78f34b06af80e6f0605edf15d5798b"}