{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:ZYG45DVMCI3RXBS7WDASKCXA53","short_pith_number":"pith:ZYG45DVM","canonical_record":{"source":{"id":"2104.13844","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-04-28T15:50:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a60670b06735cf2947007a5da3699cb7be470c20ffb3fa06a672a7dcda73850a","abstract_canon_sha256":"d6982a43633c02dddac009b40a3cfb93dbfd7df047046f9e8566bbe9270dca8e"},"schema_version":"1.0"},"canonical_sha256":"ce0dce8eac12371b865fb0c1250ae0eed0060fa42387267bad912d4d14214b37","source":{"kind":"arxiv","id":"2104.13844","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2104.13844","created_at":"2026-07-05T08:50:49Z"},{"alias_kind":"arxiv_version","alias_value":"2104.13844v3","created_at":"2026-07-05T08:50:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.13844","created_at":"2026-07-05T08:50:49Z"},{"alias_kind":"pith_short_12","alias_value":"ZYG45DVMCI3R","created_at":"2026-07-05T08:50:49Z"},{"alias_kind":"pith_short_16","alias_value":"ZYG45DVMCI3RXBS7","created_at":"2026-07-05T08:50:49Z"},{"alias_kind":"pith_short_8","alias_value":"ZYG45DVM","created_at":"2026-07-05T08:50:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:ZYG45DVMCI3RXBS7WDASKCXA53","target":"record","payload":{"canonical_record":{"source":{"id":"2104.13844","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-04-28T15:50:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a60670b06735cf2947007a5da3699cb7be470c20ffb3fa06a672a7dcda73850a","abstract_canon_sha256":"d6982a43633c02dddac009b40a3cfb93dbfd7df047046f9e8566bbe9270dca8e"},"schema_version":"1.0"},"canonical_sha256":"ce0dce8eac12371b865fb0c1250ae0eed0060fa42387267bad912d4d14214b37","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:50:49.511159Z","signature_b64":"/hNb9NFvxl5pXOJwtfAnFC0ST4oot95RTiiq2CY7cd1oLE0/T3pHi7ggQO60FANn2KxQ1GdkyTMQaOSKv4ZQBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce0dce8eac12371b865fb0c1250ae0eed0060fa42387267bad912d4d14214b37","last_reissued_at":"2026-07-05T08:50:49.510685Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:50:49.510685Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2104.13844","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:50:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PU+7a80hdZbLTpNvi14/0G7b7/PQlFIq+TiPSgJ74vpOPcpBpEcILrmuySPn4Ynd8kDohLfHIyo44XnNoSLTDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T06:04:46.981640Z"},"content_sha256":"620b4a397093f69c12201a100ce049b966015dd5c7c47769a7811fd2bee95228","schema_version":"1.0","event_id":"sha256:620b4a397093f69c12201a100ce049b966015dd5c7c47769a7811fd2bee95228"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:ZYG45DVMCI3RXBS7WDASKCXA53","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"A Generalized Projected Bellman Error for Off-policy Value Estimation in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Adam White, Andrew Patterson, Martha White","submitted_at":"2021-04-28T15:50:34Z","abstract_excerpt":"Many reinforcement learning algorithms rely on value estimation, however, the most widely used algorithms -- namely temporal difference algorithms -- can diverge under both off-policy sampling and nonlinear function approximation. Many algorithms have been developed for off-policy value estimation based on the linear mean squared projected Bellman error (MSPBE) and are sound under linear function approximation. Extending these methods to the nonlinear case has been largely unsuccessful. Recently, several methods have been introduced that approximate a different objective -- the mean-squared Be"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.13844","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.13844/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:50:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ycAzFkCYxJ2uTLAk5O4Zz7PRzBn3n1Mf2F8Ze+MgdcmvhdCQ6Cg8FWvfTPEcBvLyrWWRN75U/yGRtB5DRKknAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T06:04:46.982132Z"},"content_sha256":"78d1825afd3986058cc74a6dc26a09e29fc1f32f8c380ebd9a764cc11bd97beb","schema_version":"1.0","event_id":"sha256:78d1825afd3986058cc74a6dc26a09e29fc1f32f8c380ebd9a764cc11bd97beb"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/ZYG45DVMCI3RXBS7WDASKCXA53/bundle.json","state_url":"https://pith.science/pith/ZYG45DVMCI3RXBS7WDASKCXA53/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/ZYG45DVMCI3RXBS7WDASKCXA53/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T06:04:46Z","links":{"resolver":"https://pith.science/pith/ZYG45DVMCI3RXBS7WDASKCXA53","bundle":"https://pith.science/pith/ZYG45DVMCI3RXBS7WDASKCXA53/bundle.json","state":"https://pith.science/pith/ZYG45DVMCI3RXBS7WDASKCXA53/state.json","well_known_bundle":"https://pith.science/.well-known/pith/ZYG45DVMCI3RXBS7WDASKCXA53/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:ZYG45DVMCI3RXBS7WDASKCXA53","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d6982a43633c02dddac009b40a3cfb93dbfd7df047046f9e8566bbe9270dca8e","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-04-28T15:50:34Z","title_canon_sha256":"a60670b06735cf2947007a5da3699cb7be470c20ffb3fa06a672a7dcda73850a"},"schema_version":"1.0","source":{"id":"2104.13844","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2104.13844","created_at":"2026-07-05T08:50:49Z"},{"alias_kind":"arxiv_version","alias_value":"2104.13844v3","created_at":"2026-07-05T08:50:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.13844","created_at":"2026-07-05T08:50:49Z"},{"alias_kind":"pith_short_12","alias_value":"ZYG45DVMCI3R","created_at":"2026-07-05T08:50:49Z"},{"alias_kind":"pith_short_16","alias_value":"ZYG45DVMCI3RXBS7","created_at":"2026-07-05T08:50:49Z"},{"alias_kind":"pith_short_8","alias_value":"ZYG45DVM","created_at":"2026-07-05T08:50:49Z"}],"graph_snapshots":[{"event_id":"sha256:78d1825afd3986058cc74a6dc26a09e29fc1f32f8c380ebd9a764cc11bd97beb","target":"graph","created_at":"2026-07-05T08:50:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2104.13844/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Many reinforcement learning algorithms rely on value estimation, however, the most widely used algorithms -- namely temporal difference algorithms -- can diverge under both off-policy sampling and nonlinear function approximation. Many algorithms have been developed for off-policy value estimation based on the linear mean squared projected Bellman error (MSPBE) and are sound under linear function approximation. Extending these methods to the nonlinear case has been largely unsuccessful. Recently, several methods have been introduced that approximate a different objective -- the mean-squared Be","authors_text":"Adam White, Andrew Patterson, Martha White","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-04-28T15:50:34Z","title":"A Generalized Projected Bellman Error for Off-policy Value Estimation in Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.13844","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:620b4a397093f69c12201a100ce049b966015dd5c7c47769a7811fd2bee95228","target":"record","created_at":"2026-07-05T08:50:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d6982a43633c02dddac009b40a3cfb93dbfd7df047046f9e8566bbe9270dca8e","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-04-28T15:50:34Z","title_canon_sha256":"a60670b06735cf2947007a5da3699cb7be470c20ffb3fa06a672a7dcda73850a"},"schema_version":"1.0","source":{"id":"2104.13844","kind":"arxiv","version":3}},"canonical_sha256":"ce0dce8eac12371b865fb0c1250ae0eed0060fa42387267bad912d4d14214b37","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ce0dce8eac12371b865fb0c1250ae0eed0060fa42387267bad912d4d14214b37","first_computed_at":"2026-07-05T08:50:49.510685Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:50:49.510685Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"/hNb9NFvxl5pXOJwtfAnFC0ST4oot95RTiiq2CY7cd1oLE0/T3pHi7ggQO60FANn2KxQ1GdkyTMQaOSKv4ZQBA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:50:49.511159Z","signed_message":"canonical_sha256_bytes"},"source_id":"2104.13844","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:620b4a397093f69c12201a100ce049b966015dd5c7c47769a7811fd2bee95228","sha256:78d1825afd3986058cc74a6dc26a09e29fc1f32f8c380ebd9a764cc11bd97beb"],"state_sha256":"479e3e0e1ee0f44ef940c5788a811ccf246a9d23a8dc8d2b9fb912a5443aabd4"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"RF7PAlQIisHsCHIC7UhzkqUt0ZVXs3NjIjBc8nQcSQy0/Mvy8/zyIlwoM9SqQu2u4hiVXW0t0+6urdQqlcTMDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T06:04:46.985660Z","bundle_sha256":"2c8179b0a36d7a56d31dbc4d1074e02da72fd7e3b49d7184b5ef868b0d8de18a"}}