{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2016:S4M7E6664G52ETYEK2H2XDPWSX","short_pith_number":"pith:S4M7E666","canonical_record":{"source":{"id":"1604.06778","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2016-04-22T18:57:24Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"e4d85efe079b25bfdca29c59479dfe6c75e98edd46e5818f7fa6ea8edbc258c2","abstract_canon_sha256":"b81cbcc843c36a8e2f21744c33d92c1a92addc793f2e4c4f14a49972766ac8a1"},"schema_version":"1.0"},"canonical_sha256":"9719f27bdee1bba24f04568fab8df695d8dfd0bc01e798a47c522540b92c7d81","source":{"kind":"arxiv","id":"1604.06778","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1604.06778","created_at":"2026-05-18T01:13:30Z"},{"alias_kind":"arxiv_version","alias_value":"1604.06778v3","created_at":"2026-05-18T01:13:30Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1604.06778","created_at":"2026-05-18T01:13:30Z"},{"alias_kind":"pith_short_12","alias_value":"S4M7E6664G52","created_at":"2026-05-18T12:30:41Z"},{"alias_kind":"pith_short_16","alias_value":"S4M7E6664G52ETYE","created_at":"2026-05-18T12:30:41Z"},{"alias_kind":"pith_short_8","alias_value":"S4M7E666","created_at":"2026-05-18T12:30:41Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2016:S4M7E6664G52ETYEK2H2XDPWSX","target":"record","payload":{"canonical_record":{"source":{"id":"1604.06778","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2016-04-22T18:57:24Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"e4d85efe079b25bfdca29c59479dfe6c75e98edd46e5818f7fa6ea8edbc258c2","abstract_canon_sha256":"b81cbcc843c36a8e2f21744c33d92c1a92addc793f2e4c4f14a49972766ac8a1"},"schema_version":"1.0"},"canonical_sha256":"9719f27bdee1bba24f04568fab8df695d8dfd0bc01e798a47c522540b92c7d81","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T01:13:30.205032Z","signature_b64":"p3FNgE0MdFP/PzWJR8KqkqteMHrRSx9lUkvSAyPlreopVPvo64hmXMwVD8BSIU+7pK3gcJhEMeLFKohIn2nFDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9719f27bdee1bba24f04568fab8df695d8dfd0bc01e798a47c522540b92c7d81","last_reissued_at":"2026-05-18T01:13:30.204382Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T01:13:30.204382Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1604.06778","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T01:13:30Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"slE4wb+aEl3ClH5cgnHEyFRczpXxwE7UN1MVu5YYXmZMmSEGhg1HALmYAluLlh5XXS6w7l+VqLa3t0PgTUCNBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-31T15:29:04.672519Z"},"content_sha256":"b275a5e92f96dd57494b1948a95ea5288fdb359da37391b0cc1d2e78484c3f90","schema_version":"1.0","event_id":"sha256:b275a5e92f96dd57494b1948a95ea5288fdb359da37391b0cc1d2e78484c3f90"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2016:S4M7E6664G52ETYEK2H2XDPWSX","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Benchmarking Deep Reinforcement Learning for Continuous Control","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"John Schulman, Pieter Abbeel, Rein Houthooft, Xi Chen, Yan Duan","submitted_at":"2016-04-22T18:57:24Z","abstract_excerpt":"Recently, researchers have made significant progress combining the advances in deep learning for learning feature representations with reinforcement learning. Some notable examples include training agents to play Atari games based on raw pixel data and to acquire advanced manipulation skills using raw sensory inputs. However, it has been difficult to quantify progress in the domain of continuous control due to the lack of a commonly adopted benchmark. In this work, we present a benchmark suite of continuous control tasks, including classic tasks like cart-pole swing-up, tasks with very high st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1604.06778","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T01:13:30Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zVCxVq18LLGSyYaV52a8+uIFC7OmFtc4J/gie7y2NKYdnW4JkGWMxVEDUjrL8d96Ydkf0MXgOo4XmV5MUkWtBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-31T15:29:04.673255Z"},"content_sha256":"9b1073ccc8b33d9b6b13574d71fc9e02fa804da931ed55951b50966157d2b736","schema_version":"1.0","event_id":"sha256:9b1073ccc8b33d9b6b13574d71fc9e02fa804da931ed55951b50966157d2b736"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/S4M7E6664G52ETYEK2H2XDPWSX/bundle.json","state_url":"https://pith.science/pith/S4M7E6664G52ETYEK2H2XDPWSX/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/S4M7E6664G52ETYEK2H2XDPWSX/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-31T15:29:04Z","links":{"resolver":"https://pith.science/pith/S4M7E6664G52ETYEK2H2XDPWSX","bundle":"https://pith.science/pith/S4M7E6664G52ETYEK2H2XDPWSX/bundle.json","state":"https://pith.science/pith/S4M7E6664G52ETYEK2H2XDPWSX/state.json","well_known_bundle":"https://pith.science/.well-known/pith/S4M7E6664G52ETYEK2H2XDPWSX/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2016:S4M7E6664G52ETYEK2H2XDPWSX","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b81cbcc843c36a8e2f21744c33d92c1a92addc793f2e4c4f14a49972766ac8a1","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2016-04-22T18:57:24Z","title_canon_sha256":"e4d85efe079b25bfdca29c59479dfe6c75e98edd46e5818f7fa6ea8edbc258c2"},"schema_version":"1.0","source":{"id":"1604.06778","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1604.06778","created_at":"2026-05-18T01:13:30Z"},{"alias_kind":"arxiv_version","alias_value":"1604.06778v3","created_at":"2026-05-18T01:13:30Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1604.06778","created_at":"2026-05-18T01:13:30Z"},{"alias_kind":"pith_short_12","alias_value":"S4M7E6664G52","created_at":"2026-05-18T12:30:41Z"},{"alias_kind":"pith_short_16","alias_value":"S4M7E6664G52ETYE","created_at":"2026-05-18T12:30:41Z"},{"alias_kind":"pith_short_8","alias_value":"S4M7E666","created_at":"2026-05-18T12:30:41Z"}],"graph_snapshots":[{"event_id":"sha256:9b1073ccc8b33d9b6b13574d71fc9e02fa804da931ed55951b50966157d2b736","target":"graph","created_at":"2026-05-18T01:13:30Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"Recently, researchers have made significant progress combining the advances in deep learning for learning feature representations with reinforcement learning. Some notable examples include training agents to play Atari games based on raw pixel data and to acquire advanced manipulation skills using raw sensory inputs. However, it has been difficult to quantify progress in the domain of continuous control due to the lack of a commonly adopted benchmark. In this work, we present a benchmark suite of continuous control tasks, including classic tasks like cart-pole swing-up, tasks with very high st","authors_text":"John Schulman, Pieter Abbeel, Rein Houthooft, Xi Chen, Yan Duan","cross_cats":["cs.AI","cs.RO"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2016-04-22T18:57:24Z","title":"Benchmarking Deep Reinforcement Learning for Continuous Control"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1604.06778","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b275a5e92f96dd57494b1948a95ea5288fdb359da37391b0cc1d2e78484c3f90","target":"record","created_at":"2026-05-18T01:13:30Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b81cbcc843c36a8e2f21744c33d92c1a92addc793f2e4c4f14a49972766ac8a1","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2016-04-22T18:57:24Z","title_canon_sha256":"e4d85efe079b25bfdca29c59479dfe6c75e98edd46e5818f7fa6ea8edbc258c2"},"schema_version":"1.0","source":{"id":"1604.06778","kind":"arxiv","version":3}},"canonical_sha256":"9719f27bdee1bba24f04568fab8df695d8dfd0bc01e798a47c522540b92c7d81","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9719f27bdee1bba24f04568fab8df695d8dfd0bc01e798a47c522540b92c7d81","first_computed_at":"2026-05-18T01:13:30.204382Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T01:13:30.204382Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"p3FNgE0MdFP/PzWJR8KqkqteMHrRSx9lUkvSAyPlreopVPvo64hmXMwVD8BSIU+7pK3gcJhEMeLFKohIn2nFDw==","signature_status":"signed_v1","signed_at":"2026-05-18T01:13:30.205032Z","signed_message":"canonical_sha256_bytes"},"source_id":"1604.06778","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b275a5e92f96dd57494b1948a95ea5288fdb359da37391b0cc1d2e78484c3f90","sha256:9b1073ccc8b33d9b6b13574d71fc9e02fa804da931ed55951b50966157d2b736"],"state_sha256":"838b529e6f4dafd3baf12d8a3c56e02bcc5287fd4a5e0fb9088f8119701386ff"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"GVI7uof9y+EpTu3w2xG2GtFJn6bJZneZ3Ysu5d10bYrrkp9Cm2bMlBY5LmKtF9lWPhjAuvX2MBu/Axn04582Aw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-31T15:29:04.677764Z","bundle_sha256":"9af4f0ea5b8836252ad407a68414115a93603606e39b5c8683ed17c6bbbfc5e0"}}