{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2015:4O4WP36VK6EJ6YDVCNH5VGEUVV","short_pith_number":"pith:4O4WP36V","canonical_record":{"source":{"id":"1506.02438","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2015-06-08T11:12:48Z","cross_cats_sorted":["cs.RO","cs.SY","eess.SY"],"title_canon_sha256":"d43df1577415f60245e58f5988ab904d0630c3572cfb1a4c64c7fb158bdbcacf","abstract_canon_sha256":"6f21ada608713125c7cf1922ba3e2a5460966229a027f8096c6d834d43720cd9"},"schema_version":"1.0"},"canonical_sha256":"e3b967efd557889f6075134fda9894ad4aa21c834a4ca82756e41f604dbb760b","source":{"kind":"arxiv","id":"1506.02438","version":6},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1506.02438","created_at":"2026-06-04T17:09:19Z"},{"alias_kind":"arxiv_version","alias_value":"1506.02438v6","created_at":"2026-06-04T17:09:19Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1506.02438","created_at":"2026-06-04T17:09:19Z"},{"alias_kind":"pith_short_12","alias_value":"4O4WP36VK6EJ","created_at":"2026-06-04T17:09:19Z"},{"alias_kind":"pith_short_16","alias_value":"4O4WP36VK6EJ6YDV","created_at":"2026-06-04T17:09:19Z"},{"alias_kind":"pith_short_8","alias_value":"4O4WP36V","created_at":"2026-06-04T17:09:19Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2015:4O4WP36VK6EJ6YDVCNH5VGEUVV","target":"record","payload":{"canonical_record":{"source":{"id":"1506.02438","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2015-06-08T11:12:48Z","cross_cats_sorted":["cs.RO","cs.SY","eess.SY"],"title_canon_sha256":"d43df1577415f60245e58f5988ab904d0630c3572cfb1a4c64c7fb158bdbcacf","abstract_canon_sha256":"6f21ada608713125c7cf1922ba3e2a5460966229a027f8096c6d834d43720cd9"},"schema_version":"1.0"},"canonical_sha256":"e3b967efd557889f6075134fda9894ad4aa21c834a4ca82756e41f604dbb760b","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-04T17:09:19.470633Z","signature_b64":"6JlmxLm+iFOOmN77dEwTr9e4UZnJEhGUykfZ+mSuU3usAdovceT43KqzHrj0q47GGWPv5tVv6ck9VALtVjDADg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e3b967efd557889f6075134fda9894ad4aa21c834a4ca82756e41f604dbb760b","last_reissued_at":"2026-06-04T17:09:19.470090Z","signature_status":"signed_v1","first_computed_at":"2026-06-04T17:09:19.470090Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1506.02438","source_version":6,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-04T17:09:19Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+DEJwvbU2bgaW20UuBYqndKQ/yF4dLzJIXlLFIrEcV8mlo15OixiFc7nt1hFWaFNHmE+9ZuiFjZXD5AEnmy8Bg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T06:07:56.498995Z"},"content_sha256":"94156bacfb9a8ace7979add357de286bba36331d6a6e4befa201edd55bb38985","schema_version":"1.0","event_id":"sha256:94156bacfb9a8ace7979add357de286bba36331d6a6e4befa201edd55bb38985"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2015:4O4WP36VK6EJ6YDVCNH5VGEUVV","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"High-Dimensional Continuous Control Using Generalized Advantage Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Generalized advantage estimation reduces variance in policy gradients for high-dimensional continuous control by exponentially weighting temporal difference residuals.","cross_cats":["cs.RO","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"John Schulman, Michael Jordan, Philipp Moritz, Pieter Abbeel, Sergey Levine","submitted_at":"2015-06-08T11:12:48Z","abstract_excerpt":"Policy gradient methods are an appealing approach in reinforcement learning because they directly optimize the cumulative reward and can straightforwardly be used with nonlinear function approximators such as neural networks. The two main challenges are the large number of samples typically required, and the difficulty of obtaining stable and steady improvement despite the nonstationarity of the incoming data. We address the first challenge by using value functions to substantially reduce the variance of policy gradient estimates at the cost of some bias, with an exponentially-weighted estimat"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"We address the first challenge by using value functions to substantially reduce the variance of policy gradient estimates at the cost of some bias, with an exponentially-weighted estimator of the advantage function that is analogous to TD(lambda).","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That a neural network value function approximator can be trained sufficiently accurately to deliver useful advantage estimates without introducing bias that negates the variance reduction in high-dimensional continuous control.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Generalized advantage estimation combined with trust region optimization enables stable neural network policy learning for complex continuous control from raw kinematics.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Generalized advantage estimation reduces variance in policy gradients for high-dimensional continuous control by exponentially weighting temporal difference residuals.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"7fc55db9412099b0160703fb715343ed486b92f49b984be4dd11fe3fa77215d3"},"source":{"id":"1506.02438","kind":"arxiv","version":6},"verdict":{"id":"20115e18-0689-40a1-82cc-55ec5ea8e660","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-11T04:15:59.942358Z","strongest_claim":"We address the first challenge by using value functions to substantially reduce the variance of policy gradient estimates at the cost of some bias, with an exponentially-weighted estimator of the advantage function that is analogous to TD(lambda).","one_line_summary":"Generalized advantage estimation combined with trust region optimization enables stable neural network policy learning for complex continuous control from raw kinematics.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That a neural network value function approximator can be trained sufficiently accurately to deliver useful advantage estimates without introducing bias that negates the variance reduction in high-dimensional continuous control.","pith_extraction_headline":"Generalized advantage estimation reduces variance in policy gradients for high-dimensional continuous control by exponentially weighting temporal difference residuals."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1506.02438/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":25,"sample":[{"doi":"","year":1983,"title":"Neuronlike adaptive elements that can solve difficult learning control problems","work_id":"4aef49fd-a6b4-4a26-8fbf-a6cf90ec495c","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2000,"title":"Reinforcement learning in POMDP s via direct gradient ascent","work_id":"302c729b-2662-4394-9db1-0fe74e442094","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2012,"title":"Dynamic programming and optimal control, volume 2","work_id":"04567b70-0d02-49a4-944c-bad3be7a1cb5","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2009,"title":"Convergent temporal-difference learning with arbitrary smooth function approximation","work_id":"4ee60274-2ce7-4732-ba8f-81702324359d","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2004,"title":"Variance reduction techniques for gradient estimates in reinforcement learning","work_id":"8208938d-e1cf-4be8-9c8e-f8212d12f1dc","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":25,"snapshot_sha256":"92e22da771f03be41553c4b3ccbc30a07a507d8249b8f68d5a7cddefd4813355","internal_anchors":1},"formal_canon":{"evidence_count":1,"snapshot_sha256":"348897eecfc5c3be1e1f751b3d46c0c3e46e26edad193a1ad61ea8c6926dcc65"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"20115e18-0689-40a1-82cc-55ec5ea8e660"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-04T17:09:19Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xF4T9ak4rPcBg+h/akU93SKkrNAtgCl4ZzkmKc3OzmCrVS1HAg62sFUqXp14MJUfmCATrJuLJXzV3Fz/fhiEBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T06:07:56.499788Z"},"content_sha256":"9a96dfc1898fb120423356cf0abe2db1617cc56c6cb258cc64705c8debabf992","schema_version":"1.0","event_id":"sha256:9a96dfc1898fb120423356cf0abe2db1617cc56c6cb258cc64705c8debabf992"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/4O4WP36VK6EJ6YDVCNH5VGEUVV/bundle.json","state_url":"https://pith.science/pith/4O4WP36VK6EJ6YDVCNH5VGEUVV/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/4O4WP36VK6EJ6YDVCNH5VGEUVV/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-18T06:07:56Z","links":{"resolver":"https://pith.science/pith/4O4WP36VK6EJ6YDVCNH5VGEUVV","bundle":"https://pith.science/pith/4O4WP36VK6EJ6YDVCNH5VGEUVV/bundle.json","state":"https://pith.science/pith/4O4WP36VK6EJ6YDVCNH5VGEUVV/state.json","well_known_bundle":"https://pith.science/.well-known/pith/4O4WP36VK6EJ6YDVCNH5VGEUVV/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2015:4O4WP36VK6EJ6YDVCNH5VGEUVV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6f21ada608713125c7cf1922ba3e2a5460966229a027f8096c6d834d43720cd9","cross_cats_sorted":["cs.RO","cs.SY","eess.SY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2015-06-08T11:12:48Z","title_canon_sha256":"d43df1577415f60245e58f5988ab904d0630c3572cfb1a4c64c7fb158bdbcacf"},"schema_version":"1.0","source":{"id":"1506.02438","kind":"arxiv","version":6}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1506.02438","created_at":"2026-06-04T17:09:19Z"},{"alias_kind":"arxiv_version","alias_value":"1506.02438v6","created_at":"2026-06-04T17:09:19Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1506.02438","created_at":"2026-06-04T17:09:19Z"},{"alias_kind":"pith_short_12","alias_value":"4O4WP36VK6EJ","created_at":"2026-06-04T17:09:19Z"},{"alias_kind":"pith_short_16","alias_value":"4O4WP36VK6EJ6YDV","created_at":"2026-06-04T17:09:19Z"},{"alias_kind":"pith_short_8","alias_value":"4O4WP36V","created_at":"2026-06-04T17:09:19Z"}],"graph_snapshots":[{"event_id":"sha256:9a96dfc1898fb120423356cf0abe2db1617cc56c6cb258cc64705c8debabf992","target":"graph","created_at":"2026-06-04T17:09:19Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"We address the first challenge by using value functions to substantially reduce the variance of policy gradient estimates at the cost of some bias, with an exponentially-weighted estimator of the advantage function that is analogous to TD(lambda)."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That a neural network value function approximator can be trained sufficiently accurately to deliver useful advantage estimates without introducing bias that negates the variance reduction in high-dimensional continuous control."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Generalized advantage estimation combined with trust region optimization enables stable neural network policy learning for complex continuous control from raw kinematics."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Generalized advantage estimation reduces variance in policy gradients for high-dimensional continuous control by exponentially weighting temporal difference residuals."}],"snapshot_sha256":"7fc55db9412099b0160703fb715343ed486b92f49b984be4dd11fe3fa77215d3"},"formal_canon":{"evidence_count":1,"snapshot_sha256":"348897eecfc5c3be1e1f751b3d46c0c3e46e26edad193a1ad61ea8c6926dcc65"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1506.02438/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Policy gradient methods are an appealing approach in reinforcement learning because they directly optimize the cumulative reward and can straightforwardly be used with nonlinear function approximators such as neural networks. The two main challenges are the large number of samples typically required, and the difficulty of obtaining stable and steady improvement despite the nonstationarity of the incoming data. We address the first challenge by using value functions to substantially reduce the variance of policy gradient estimates at the cost of some bias, with an exponentially-weighted estimat","authors_text":"John Schulman, Michael Jordan, Philipp Moritz, Pieter Abbeel, Sergey Levine","cross_cats":["cs.RO","cs.SY","eess.SY"],"headline":"Generalized advantage estimation reduces variance in policy gradients for high-dimensional continuous control by exponentially weighting temporal difference residuals.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2015-06-08T11:12:48Z","title":"High-Dimensional Continuous Control Using Generalized Advantage Estimation"},"references":{"count":25,"internal_anchors":1,"resolved_work":25,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"Neuronlike adaptive elements that can solve difficult learning control problems","work_id":"4aef49fd-a6b4-4a26-8fbf-a6cf90ec495c","year":1983},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"Reinforcement learning in POMDP s via direct gradient ascent","work_id":"302c729b-2662-4394-9db1-0fe74e442094","year":2000},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Dynamic programming and optimal control, volume 2","work_id":"04567b70-0d02-49a4-944c-bad3be7a1cb5","year":2012},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":4,"title":"Convergent temporal-difference learning with arbitrary smooth function approximation","work_id":"4ee60274-2ce7-4732-ba8f-81702324359d","year":2009},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Variance reduction techniques for gradient estimates in reinforcement learning","work_id":"8208938d-e1cf-4be8-9c8e-f8212d12f1dc","year":2004}],"snapshot_sha256":"92e22da771f03be41553c4b3ccbc30a07a507d8249b8f68d5a7cddefd4813355"},"source":{"id":"1506.02438","kind":"arxiv","version":6},"verdict":{"created_at":"2026-05-11T04:15:59.942358Z","id":"20115e18-0689-40a1-82cc-55ec5ea8e660","model_set":{"reader":"grok-4.3"},"one_line_summary":"Generalized advantage estimation combined with trust region optimization enables stable neural network policy learning for complex continuous control from raw kinematics.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Generalized advantage estimation reduces variance in policy gradients for high-dimensional continuous control by exponentially weighting temporal difference residuals.","strongest_claim":"We address the first challenge by using value functions to substantially reduce the variance of policy gradient estimates at the cost of some bias, with an exponentially-weighted estimator of the advantage function that is analogous to TD(lambda).","weakest_assumption":"That a neural network value function approximator can be trained sufficiently accurately to deliver useful advantage estimates without introducing bias that negates the variance reduction in high-dimensional continuous control."}},"verdict_id":"20115e18-0689-40a1-82cc-55ec5ea8e660"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:94156bacfb9a8ace7979add357de286bba36331d6a6e4befa201edd55bb38985","target":"record","created_at":"2026-06-04T17:09:19Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6f21ada608713125c7cf1922ba3e2a5460966229a027f8096c6d834d43720cd9","cross_cats_sorted":["cs.RO","cs.SY","eess.SY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2015-06-08T11:12:48Z","title_canon_sha256":"d43df1577415f60245e58f5988ab904d0630c3572cfb1a4c64c7fb158bdbcacf"},"schema_version":"1.0","source":{"id":"1506.02438","kind":"arxiv","version":6}},"canonical_sha256":"e3b967efd557889f6075134fda9894ad4aa21c834a4ca82756e41f604dbb760b","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e3b967efd557889f6075134fda9894ad4aa21c834a4ca82756e41f604dbb760b","first_computed_at":"2026-06-04T17:09:19.470090Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-04T17:09:19.470090Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"6JlmxLm+iFOOmN77dEwTr9e4UZnJEhGUykfZ+mSuU3usAdovceT43KqzHrj0q47GGWPv5tVv6ck9VALtVjDADg==","signature_status":"signed_v1","signed_at":"2026-06-04T17:09:19.470633Z","signed_message":"canonical_sha256_bytes"},"source_id":"1506.02438","source_kind":"arxiv","source_version":6}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:94156bacfb9a8ace7979add357de286bba36331d6a6e4befa201edd55bb38985","sha256:9a96dfc1898fb120423356cf0abe2db1617cc56c6cb258cc64705c8debabf992"],"state_sha256":"a8013dd11057214adfbe7e33dcdb447bd6a729e274598b5881acc46a3717554f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"1x7y17hy/yj/6qIP8QHulbXy5zz7RqFtwaHwJ7ZrO2tLCNDA2qACAH6IBxsLPZHt7AIFQcfajfMVr+tbenp8DQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-18T06:07:56.505664Z","bundle_sha256":"1f83e81630d5a7042e70c5185314645dd0f720ba262c35ea63d4f643f1a2f5b7"}}