{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2020:VYFDDXJ4ATVA6LDKSOZOF2KRYK","short_pith_number":"pith:VYFDDXJ4","canonical_record":{"source":{"id":"2001.06781","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-01-19T06:07:20Z","cross_cats_sorted":[],"title_canon_sha256":"d3ec8e4ab44f325ba5c9777938757ad1b0b3403fe813f659ff7a005eeabefccf","abstract_canon_sha256":"327aaee09b7c3c5d38c8c6f6c0d291136847c6c294efacb6ac7256f289c685e8"},"schema_version":"1.0"},"canonical_sha256":"ae0a31dd3c04ea0f2c6a93b2e2e951c2a8cc8c91af0ccf92090600d7994c1094","source":{"kind":"arxiv","id":"2001.06781","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2001.06781","created_at":"2026-07-05T00:34:24Z"},{"alias_kind":"arxiv_version","alias_value":"2001.06781v1","created_at":"2026-07-05T00:34:24Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2001.06781","created_at":"2026-07-05T00:34:24Z"},{"alias_kind":"pith_short_12","alias_value":"VYFDDXJ4ATVA","created_at":"2026-07-05T00:34:24Z"},{"alias_kind":"pith_short_16","alias_value":"VYFDDXJ4ATVA6LDK","created_at":"2026-07-05T00:34:24Z"},{"alias_kind":"pith_short_8","alias_value":"VYFDDXJ4","created_at":"2026-07-05T00:34:24Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2020:VYFDDXJ4ATVA6LDKSOZOF2KRYK","target":"record","payload":{"canonical_record":{"source":{"id":"2001.06781","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-01-19T06:07:20Z","cross_cats_sorted":[],"title_canon_sha256":"d3ec8e4ab44f325ba5c9777938757ad1b0b3403fe813f659ff7a005eeabefccf","abstract_canon_sha256":"327aaee09b7c3c5d38c8c6f6c0d291136847c6c294efacb6ac7256f289c685e8"},"schema_version":"1.0"},"canonical_sha256":"ae0a31dd3c04ea0f2c6a93b2e2e951c2a8cc8c91af0ccf92090600d7994c1094","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:34:24.644522Z","signature_b64":"UTCw/owixpd80vliwHZEGp+m1WQpc8bvcgtcWwF4xAHG5HeAKIJvn6yoo3R9U5LF2i9dGEqq1p8SBGfxdPPfBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae0a31dd3c04ea0f2c6a93b2e2e951c2a8cc8c91af0ccf92090600d7994c1094","last_reissued_at":"2026-07-05T00:34:24.644120Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:34:24.644120Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2001.06781","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:34:24Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4V52jgeJkR30Ffm5fLclciTtiXDoEfdAyNufD9LP53gAZIFtws53e1SgCemdYtVFrbOq9Kw/+1OiQS3ekQ9JAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T10:18:01.352039Z"},"content_sha256":"0ecddb5ff61bb00ca40f147d283fe23a02f7b3896586e852753889d4e74d43d2","schema_version":"1.0","event_id":"sha256:0ecddb5ff61bb00ca40f147d283fe23a02f7b3896586e852753889d4e74d43d2"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2020:VYFDDXJ4ATVA6LDKSOZOF2KRYK","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"FRESH: Interactive Reward Shaping in High-Dimensional State Spaces using Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Andrew Clark, Baicen Xiao, Bhaskar Ramasubramanian, Linda Bushnell, Qifan Lu, Radha Poovendran","submitted_at":"2020-01-19T06:07:20Z","abstract_excerpt":"Reinforcement learning has been successful in training autonomous agents to accomplish goals in complex environments. Although this has been adapted to multiple settings, including robotics and computer games, human players often find it easier to obtain higher rewards in some environments than reinforcement learning algorithms. This is especially true of high-dimensional state spaces where the reward obtained by the agent is sparse or extremely delayed. In this paper, we seek to effectively integrate feedback signals supplied by a human operator with deep reinforcement learning algorithms in "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2001.06781","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2001.06781/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:34:24Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dverl88GxeCYufJsLMSEMZqveFIIPE9ynOnwM9kSJ/gvj8HpNqELsEprwpenY3Yuab7pQoSd82F671EISx0/CQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T10:18:01.352550Z"},"content_sha256":"fa40c7b2d21127df209af211f97db1c8843a10543b9b55fe854a3fe2bc7e9e7f","schema_version":"1.0","event_id":"sha256:fa40c7b2d21127df209af211f97db1c8843a10543b9b55fe854a3fe2bc7e9e7f"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/bundle.json","state_url":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T10:18:01Z","links":{"resolver":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK","bundle":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/bundle.json","state":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/state.json","well_known_bundle":"https://pith.science/.well-known/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:VYFDDXJ4ATVA6LDKSOZOF2KRYK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"327aaee09b7c3c5d38c8c6f6c0d291136847c6c294efacb6ac7256f289c685e8","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-01-19T06:07:20Z","title_canon_sha256":"d3ec8e4ab44f325ba5c9777938757ad1b0b3403fe813f659ff7a005eeabefccf"},"schema_version":"1.0","source":{"id":"2001.06781","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2001.06781","created_at":"2026-07-05T00:34:24Z"},{"alias_kind":"arxiv_version","alias_value":"2001.06781v1","created_at":"2026-07-05T00:34:24Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2001.06781","created_at":"2026-07-05T00:34:24Z"},{"alias_kind":"pith_short_12","alias_value":"VYFDDXJ4ATVA","created_at":"2026-07-05T00:34:24Z"},{"alias_kind":"pith_short_16","alias_value":"VYFDDXJ4ATVA6LDK","created_at":"2026-07-05T00:34:24Z"},{"alias_kind":"pith_short_8","alias_value":"VYFDDXJ4","created_at":"2026-07-05T00:34:24Z"}],"graph_snapshots":[{"event_id":"sha256:fa40c7b2d21127df209af211f97db1c8843a10543b9b55fe854a3fe2bc7e9e7f","target":"graph","created_at":"2026-07-05T00:34:24Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2001.06781/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning has been successful in training autonomous agents to accomplish goals in complex environments. Although this has been adapted to multiple settings, including robotics and computer games, human players often find it easier to obtain higher rewards in some environments than reinforcement learning algorithms. This is especially true of high-dimensional state spaces where the reward obtained by the agent is sparse or extremely delayed. In this paper, we seek to effectively integrate feedback signals supplied by a human operator with deep reinforcement learning algorithms in ","authors_text":"Andrew Clark, Baicen Xiao, Bhaskar Ramasubramanian, Linda Bushnell, Qifan Lu, Radha Poovendran","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-01-19T06:07:20Z","title":"FRESH: Interactive Reward Shaping in High-Dimensional State Spaces using Human Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2001.06781","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:0ecddb5ff61bb00ca40f147d283fe23a02f7b3896586e852753889d4e74d43d2","target":"record","created_at":"2026-07-05T00:34:24Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"327aaee09b7c3c5d38c8c6f6c0d291136847c6c294efacb6ac7256f289c685e8","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-01-19T06:07:20Z","title_canon_sha256":"d3ec8e4ab44f325ba5c9777938757ad1b0b3403fe813f659ff7a005eeabefccf"},"schema_version":"1.0","source":{"id":"2001.06781","kind":"arxiv","version":1}},"canonical_sha256":"ae0a31dd3c04ea0f2c6a93b2e2e951c2a8cc8c91af0ccf92090600d7994c1094","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ae0a31dd3c04ea0f2c6a93b2e2e951c2a8cc8c91af0ccf92090600d7994c1094","first_computed_at":"2026-07-05T00:34:24.644120Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T00:34:24.644120Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"UTCw/owixpd80vliwHZEGp+m1WQpc8bvcgtcWwF4xAHG5HeAKIJvn6yoo3R9U5LF2i9dGEqq1p8SBGfxdPPfBw==","signature_status":"signed_v1","signed_at":"2026-07-05T00:34:24.644522Z","signed_message":"canonical_sha256_bytes"},"source_id":"2001.06781","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:0ecddb5ff61bb00ca40f147d283fe23a02f7b3896586e852753889d4e74d43d2","sha256:fa40c7b2d21127df209af211f97db1c8843a10543b9b55fe854a3fe2bc7e9e7f"],"state_sha256":"210c8e1e89f1aa89f3a99a6466be79f1854eb16ffe6abb7610dc24a1334691b4"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4owF5nT7RpIFrXgMPrYQ95wiEVjmsOf13YKW4849R0rOI2fD+YK1XZCw878Za/JliI6PlhnIUwdydS5Il61GDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T10:18:01.357796Z","bundle_sha256":"26b5843702b27dab6f514bde73bdcd9b952def44125a9fc1875ea5e9e2137624"}}