{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:ZNMFHG6INFKXVVU3ENUKA4RIX2","short_pith_number":"pith:ZNMFHG6I","canonical_record":{"source":{"id":"2502.01307","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T12:32:50Z","cross_cats_sorted":[],"title_canon_sha256":"367d86c319cb15ac3aad2eb7cb80b24486f979a167fd33c47c50c8e0d5054ee2","abstract_canon_sha256":"4b30dc573f7d4949b31d1e1769cc3ebf3ca40c9487f76e886ffa1c64be946c30"},"schema_version":"1.0"},"canonical_sha256":"cb58539bc869557ad69b2368a07228be9c8bdf5a82a693dee47d216a297bcebf","source":{"kind":"arxiv","id":"2502.01307","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.01307","created_at":"2026-07-05T10:08:45Z"},{"alias_kind":"arxiv_version","alias_value":"2502.01307v1","created_at":"2026-07-05T10:08:45Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01307","created_at":"2026-07-05T10:08:45Z"},{"alias_kind":"pith_short_12","alias_value":"ZNMFHG6INFKX","created_at":"2026-07-05T10:08:45Z"},{"alias_kind":"pith_short_16","alias_value":"ZNMFHG6INFKXVVU3","created_at":"2026-07-05T10:08:45Z"},{"alias_kind":"pith_short_8","alias_value":"ZNMFHG6I","created_at":"2026-07-05T10:08:45Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:ZNMFHG6INFKXVVU3ENUKA4RIX2","target":"record","payload":{"canonical_record":{"source":{"id":"2502.01307","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T12:32:50Z","cross_cats_sorted":[],"title_canon_sha256":"367d86c319cb15ac3aad2eb7cb80b24486f979a167fd33c47c50c8e0d5054ee2","abstract_canon_sha256":"4b30dc573f7d4949b31d1e1769cc3ebf3ca40c9487f76e886ffa1c64be946c30"},"schema_version":"1.0"},"canonical_sha256":"cb58539bc869557ad69b2368a07228be9c8bdf5a82a693dee47d216a297bcebf","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:45.609424Z","signature_b64":"e2E/bdmaMff2i1SSY4cCaSVv2FpxoXfIR/AaMOpfbm/li/ZJkr09MGOfZcz7sXMLNdkX2poXxUkNchywKQBDCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb58539bc869557ad69b2368a07228be9c8bdf5a82a693dee47d216a297bcebf","last_reissued_at":"2026-07-05T10:08:45.609014Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:45.609014Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.01307","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:08:45Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2bHqpEDVIxg4qHHWJdI1bNMkBF/0HF+tjlgFC3RN8qof4UK3xTgIDkJF5VfpGbqJt5lZZ4LtHgW8JwQwiNhnDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T01:56:49.513724Z"},"content_sha256":"1fc682b8d1fe2393ad052cdf9addf41682c1bd3074c0c1d18bfb3e0f1f29832a","schema_version":"1.0","event_id":"sha256:1fc682b8d1fe2393ad052cdf9addf41682c1bd3074c0c1d18bfb3e0f1f29832a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:ZNMFHG6INFKXVVU3ENUKA4RIX2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Improving the Effectiveness of Potential-Based Reward Shaping in Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Daniel Kudenko, Henrik M\\\"uller","submitted_at":"2025-02-03T12:32:50Z","abstract_excerpt":"Potential-based reward shaping is commonly used to incorporate prior knowledge of how to solve the task into reinforcement learning because it can formally guarantee policy invariance. As such, the optimal policy and the ordering of policies by their returns are not altered by potential-based reward shaping. In this work, we highlight the dependence of effective potential-based reward shaping on the initial Q-values and external rewards, which determine the agent's ability to exploit the shaping rewards to guide its exploration and achieve increased sample efficiency. We formally derive how a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01307","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01307/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:08:45Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Qd9SFbaG+jNSnU76DJvIWyYlWJgz4yw7o0oXZ5Wzp1SpyAkiWEuTVktn/4zmi+lnNLuO68D9ZdNub10Z3w3cAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T01:56:49.514111Z"},"content_sha256":"2570d763cd021c83ac8ecd642a9492979c5aa32cfc8c10c81747e4228f7edccf","schema_version":"1.0","event_id":"sha256:2570d763cd021c83ac8ecd642a9492979c5aa32cfc8c10c81747e4228f7edccf"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/ZNMFHG6INFKXVVU3ENUKA4RIX2/bundle.json","state_url":"https://pith.science/pith/ZNMFHG6INFKXVVU3ENUKA4RIX2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/ZNMFHG6INFKXVVU3ENUKA4RIX2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-13T01:56:49Z","links":{"resolver":"https://pith.science/pith/ZNMFHG6INFKXVVU3ENUKA4RIX2","bundle":"https://pith.science/pith/ZNMFHG6INFKXVVU3ENUKA4RIX2/bundle.json","state":"https://pith.science/pith/ZNMFHG6INFKXVVU3ENUKA4RIX2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/ZNMFHG6INFKXVVU3ENUKA4RIX2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:ZNMFHG6INFKXVVU3ENUKA4RIX2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4b30dc573f7d4949b31d1e1769cc3ebf3ca40c9487f76e886ffa1c64be946c30","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T12:32:50Z","title_canon_sha256":"367d86c319cb15ac3aad2eb7cb80b24486f979a167fd33c47c50c8e0d5054ee2"},"schema_version":"1.0","source":{"id":"2502.01307","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.01307","created_at":"2026-07-05T10:08:45Z"},{"alias_kind":"arxiv_version","alias_value":"2502.01307v1","created_at":"2026-07-05T10:08:45Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01307","created_at":"2026-07-05T10:08:45Z"},{"alias_kind":"pith_short_12","alias_value":"ZNMFHG6INFKX","created_at":"2026-07-05T10:08:45Z"},{"alias_kind":"pith_short_16","alias_value":"ZNMFHG6INFKXVVU3","created_at":"2026-07-05T10:08:45Z"},{"alias_kind":"pith_short_8","alias_value":"ZNMFHG6I","created_at":"2026-07-05T10:08:45Z"}],"graph_snapshots":[{"event_id":"sha256:2570d763cd021c83ac8ecd642a9492979c5aa32cfc8c10c81747e4228f7edccf","target":"graph","created_at":"2026-07-05T10:08:45Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.01307/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Potential-based reward shaping is commonly used to incorporate prior knowledge of how to solve the task into reinforcement learning because it can formally guarantee policy invariance. As such, the optimal policy and the ordering of policies by their returns are not altered by potential-based reward shaping. In this work, we highlight the dependence of effective potential-based reward shaping on the initial Q-values and external rewards, which determine the agent's ability to exploit the shaping rewards to guide its exploration and achieve increased sample efficiency. We formally derive how a ","authors_text":"Daniel Kudenko, Henrik M\\\"uller","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T12:32:50Z","title":"Improving the Effectiveness of Potential-Based Reward Shaping in Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01307","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1fc682b8d1fe2393ad052cdf9addf41682c1bd3074c0c1d18bfb3e0f1f29832a","target":"record","created_at":"2026-07-05T10:08:45Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4b30dc573f7d4949b31d1e1769cc3ebf3ca40c9487f76e886ffa1c64be946c30","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T12:32:50Z","title_canon_sha256":"367d86c319cb15ac3aad2eb7cb80b24486f979a167fd33c47c50c8e0d5054ee2"},"schema_version":"1.0","source":{"id":"2502.01307","kind":"arxiv","version":1}},"canonical_sha256":"cb58539bc869557ad69b2368a07228be9c8bdf5a82a693dee47d216a297bcebf","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"cb58539bc869557ad69b2368a07228be9c8bdf5a82a693dee47d216a297bcebf","first_computed_at":"2026-07-05T10:08:45.609014Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:08:45.609014Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"e2E/bdmaMff2i1SSY4cCaSVv2FpxoXfIR/AaMOpfbm/li/ZJkr09MGOfZcz7sXMLNdkX2poXxUkNchywKQBDCA==","signature_status":"signed_v1","signed_at":"2026-07-05T10:08:45.609424Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.01307","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1fc682b8d1fe2393ad052cdf9addf41682c1bd3074c0c1d18bfb3e0f1f29832a","sha256:2570d763cd021c83ac8ecd642a9492979c5aa32cfc8c10c81747e4228f7edccf"],"state_sha256":"525cff93a4c220e1c4fd9340af772f27357a293dcbd64e7abb71af094233ad35"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"twtVfvumQLyY0m1rVnKKxqW7oF/EaM7cra2tQjvd7kSVDaMUm9g2yX8uYO6odraUBBtG76ipQU56DV3HH/jaDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-13T01:56:49.516797Z","bundle_sha256":"f5176e2380a6adf2fe0f9a2e014224b3052e8ccf5604d08c589591236d52ccee"}}