{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2020:7NFLPGBZUHQNA5T5BBFQAWV77W","short_pith_number":"pith:7NFLPGBZ","canonical_record":{"source":{"id":"2004.00857","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-02T08:05:18Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"132e647664f497cf91c79286581fa9d41e7705b07c6e765cd0842070f9d9825c","abstract_canon_sha256":"3bcc3b9d31c9b60a1f0c9a0b953c2012a119086eebe36217162f831ab8fa03b0"},"schema_version":"1.0"},"canonical_sha256":"fb4ab79839a1e0d0767d084b005abffd812356a18347282d8e4479a9a5896fd7","source":{"kind":"arxiv","id":"2004.00857","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2004.00857","created_at":"2026-07-05T00:52:21Z"},{"alias_kind":"arxiv_version","alias_value":"2004.00857v1","created_at":"2026-07-05T00:52:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.00857","created_at":"2026-07-05T00:52:21Z"},{"alias_kind":"pith_short_12","alias_value":"7NFLPGBZUHQN","created_at":"2026-07-05T00:52:21Z"},{"alias_kind":"pith_short_16","alias_value":"7NFLPGBZUHQNA5T5","created_at":"2026-07-05T00:52:21Z"},{"alias_kind":"pith_short_8","alias_value":"7NFLPGBZ","created_at":"2026-07-05T00:52:21Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2020:7NFLPGBZUHQNA5T5BBFQAWV77W","target":"record","payload":{"canonical_record":{"source":{"id":"2004.00857","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-02T08:05:18Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"132e647664f497cf91c79286581fa9d41e7705b07c6e765cd0842070f9d9825c","abstract_canon_sha256":"3bcc3b9d31c9b60a1f0c9a0b953c2012a119086eebe36217162f831ab8fa03b0"},"schema_version":"1.0"},"canonical_sha256":"fb4ab79839a1e0d0767d084b005abffd812356a18347282d8e4479a9a5896fd7","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:52:21.291011Z","signature_b64":"TtSzY1I8EKT6j7wpd3xq/MtgElr/0jawcwgsP+K42tMRo+v46tYHD+YzGAm3bQ38WtIFMSPRYdO0IwHguyJhBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb4ab79839a1e0d0767d084b005abffd812356a18347282d8e4479a9a5896fd7","last_reissued_at":"2026-07-05T00:52:21.290680Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:52:21.290680Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2004.00857","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:52:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"wzg9MaFfAVy/OzG47O6IrglbdiIHCPWrxeUp7+MHan0IKMAioHf5H0AtO9vXKTvIQ8IAQTf+sHAaAu0Mo/kVCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T11:20:22.664854Z"},"content_sha256":"67e5479a6e4ca53ba11b518430e9e06e8c1bbf4cd2d6b37f7dd2780c2e642add","schema_version":"1.0","event_id":"sha256:67e5479a6e4ca53ba11b518430e9e06e8c1bbf4cd2d6b37f7dd2780c2e642add"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2020:7NFLPGBZUHQNA5T5BBFQAWV77W","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Average Reward Adjusted Discounted Reinforcement Learning: Near-Blackwell-Optimal Policies for Real-World Applications","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Manuel Schneckenreither","submitted_at":"2020-04-02T08:05:18Z","abstract_excerpt":"Although in recent years reinforcement learning has become very popular the number of successful applications to different kinds of operations research problems is rather scarce. Reinforcement learning is based on the well-studied dynamic programming technique and thus also aims at finding the best stationary policy for a given Markov Decision Process, but in contrast does not require any model knowledge. The policy is assessed solely on consecutive states (or state-action pairs), which are observed while an agent explores the solution space. The contributions of this paper are manifold. First"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.00857","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.00857/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:52:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"hJdGocA3KRSorVi8QTFBLGAlMZNHMjMyrRyurlQySW1efy3RPLzLm1ngsovHL2sCI+cEPxu/zOZPBUlp8erHAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T11:20:22.665246Z"},"content_sha256":"a89997e520553318a2aa4ddfcd7fc5c0d109a7079ce96a0645a108fee751fcf8","schema_version":"1.0","event_id":"sha256:a89997e520553318a2aa4ddfcd7fc5c0d109a7079ce96a0645a108fee751fcf8"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/7NFLPGBZUHQNA5T5BBFQAWV77W/bundle.json","state_url":"https://pith.science/pith/7NFLPGBZUHQNA5T5BBFQAWV77W/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/7NFLPGBZUHQNA5T5BBFQAWV77W/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T11:20:22Z","links":{"resolver":"https://pith.science/pith/7NFLPGBZUHQNA5T5BBFQAWV77W","bundle":"https://pith.science/pith/7NFLPGBZUHQNA5T5BBFQAWV77W/bundle.json","state":"https://pith.science/pith/7NFLPGBZUHQNA5T5BBFQAWV77W/state.json","well_known_bundle":"https://pith.science/.well-known/pith/7NFLPGBZUHQNA5T5BBFQAWV77W/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:7NFLPGBZUHQNA5T5BBFQAWV77W","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"3bcc3b9d31c9b60a1f0c9a0b953c2012a119086eebe36217162f831ab8fa03b0","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-02T08:05:18Z","title_canon_sha256":"132e647664f497cf91c79286581fa9d41e7705b07c6e765cd0842070f9d9825c"},"schema_version":"1.0","source":{"id":"2004.00857","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2004.00857","created_at":"2026-07-05T00:52:21Z"},{"alias_kind":"arxiv_version","alias_value":"2004.00857v1","created_at":"2026-07-05T00:52:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.00857","created_at":"2026-07-05T00:52:21Z"},{"alias_kind":"pith_short_12","alias_value":"7NFLPGBZUHQN","created_at":"2026-07-05T00:52:21Z"},{"alias_kind":"pith_short_16","alias_value":"7NFLPGBZUHQNA5T5","created_at":"2026-07-05T00:52:21Z"},{"alias_kind":"pith_short_8","alias_value":"7NFLPGBZ","created_at":"2026-07-05T00:52:21Z"}],"graph_snapshots":[{"event_id":"sha256:a89997e520553318a2aa4ddfcd7fc5c0d109a7079ce96a0645a108fee751fcf8","target":"graph","created_at":"2026-07-05T00:52:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2004.00857/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Although in recent years reinforcement learning has become very popular the number of successful applications to different kinds of operations research problems is rather scarce. Reinforcement learning is based on the well-studied dynamic programming technique and thus also aims at finding the best stationary policy for a given Markov Decision Process, but in contrast does not require any model knowledge. The policy is assessed solely on consecutive states (or state-action pairs), which are observed while an agent explores the solution space. The contributions of this paper are manifold. First","authors_text":"Manuel Schneckenreither","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-02T08:05:18Z","title":"Average Reward Adjusted Discounted Reinforcement Learning: Near-Blackwell-Optimal Policies for Real-World Applications"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.00857","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:67e5479a6e4ca53ba11b518430e9e06e8c1bbf4cd2d6b37f7dd2780c2e642add","target":"record","created_at":"2026-07-05T00:52:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"3bcc3b9d31c9b60a1f0c9a0b953c2012a119086eebe36217162f831ab8fa03b0","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-02T08:05:18Z","title_canon_sha256":"132e647664f497cf91c79286581fa9d41e7705b07c6e765cd0842070f9d9825c"},"schema_version":"1.0","source":{"id":"2004.00857","kind":"arxiv","version":1}},"canonical_sha256":"fb4ab79839a1e0d0767d084b005abffd812356a18347282d8e4479a9a5896fd7","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"fb4ab79839a1e0d0767d084b005abffd812356a18347282d8e4479a9a5896fd7","first_computed_at":"2026-07-05T00:52:21.290680Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T00:52:21.290680Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"TtSzY1I8EKT6j7wpd3xq/MtgElr/0jawcwgsP+K42tMRo+v46tYHD+YzGAm3bQ38WtIFMSPRYdO0IwHguyJhBA==","signature_status":"signed_v1","signed_at":"2026-07-05T00:52:21.291011Z","signed_message":"canonical_sha256_bytes"},"source_id":"2004.00857","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:67e5479a6e4ca53ba11b518430e9e06e8c1bbf4cd2d6b37f7dd2780c2e642add","sha256:a89997e520553318a2aa4ddfcd7fc5c0d109a7079ce96a0645a108fee751fcf8"],"state_sha256":"305226e0990e304e4986bed59f542cc792d8c733485cdc414ec2f4670857be51"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5aPA7sDpHDATZLkb6xK+c+ae9yUj/vFFPLiEqomRGK2qsYHPQmAAXuyGVhWD8Wm5PY1gdrsBpai5OvfiBgVICQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T11:20:22.668055Z","bundle_sha256":"3f400e2fa141a448ac4c0fe16dc9b57d9077bee77d864136bdbdc9882b350274"}}