{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:VYFDDXJ4ATVA6LDKSOZOF2KRYK","short_pith_number":"pith:VYFDDXJ4","schema_version":"1.0","canonical_sha256":"ae0a31dd3c04ea0f2c6a93b2e2e951c2a8cc8c91af0ccf92090600d7994c1094","source":{"kind":"arxiv","id":"2001.06781","version":1},"attestation_state":"computed","paper":{"title":"FRESH: Interactive Reward Shaping in High-Dimensional State Spaces using Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Andrew Clark, Baicen Xiao, Bhaskar Ramasubramanian, Linda Bushnell, Qifan Lu, Radha Poovendran","submitted_at":"2020-01-19T06:07:20Z","abstract_excerpt":"Reinforcement learning has been successful in training autonomous agents to accomplish goals in complex environments. Although this has been adapted to multiple settings, including robotics and computer games, human players often find it easier to obtain higher rewards in some environments than reinforcement learning algorithms. This is especially true of high-dimensional state spaces where the reward obtained by the agent is sparse or extremely delayed. In this paper, we seek to effectively integrate feedback signals supplied by a human operator with deep reinforcement learning algorithms in "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2001.06781","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-01-19T06:07:20Z","cross_cats_sorted":[],"title_canon_sha256":"d3ec8e4ab44f325ba5c9777938757ad1b0b3403fe813f659ff7a005eeabefccf","abstract_canon_sha256":"327aaee09b7c3c5d38c8c6f6c0d291136847c6c294efacb6ac7256f289c685e8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:34:24.644522Z","signature_b64":"UTCw/owixpd80vliwHZEGp+m1WQpc8bvcgtcWwF4xAHG5HeAKIJvn6yoo3R9U5LF2i9dGEqq1p8SBGfxdPPfBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae0a31dd3c04ea0f2c6a93b2e2e951c2a8cc8c91af0ccf92090600d7994c1094","last_reissued_at":"2026-07-05T00:34:24.644120Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:34:24.644120Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FRESH: Interactive Reward Shaping in High-Dimensional State Spaces using Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Andrew Clark, Baicen Xiao, Bhaskar Ramasubramanian, Linda Bushnell, Qifan Lu, Radha Poovendran","submitted_at":"2020-01-19T06:07:20Z","abstract_excerpt":"Reinforcement learning has been successful in training autonomous agents to accomplish goals in complex environments. Although this has been adapted to multiple settings, including robotics and computer games, human players often find it easier to obtain higher rewards in some environments than reinforcement learning algorithms. This is especially true of high-dimensional state spaces where the reward obtained by the agent is sparse or extremely delayed. In this paper, we seek to effectively integrate feedback signals supplied by a human operator with deep reinforcement learning algorithms in "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2001.06781","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2001.06781/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2001.06781","created_at":"2026-07-05T00:34:24.644179+00:00"},{"alias_kind":"arxiv_version","alias_value":"2001.06781v1","created_at":"2026-07-05T00:34:24.644179+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2001.06781","created_at":"2026-07-05T00:34:24.644179+00:00"},{"alias_kind":"pith_short_12","alias_value":"VYFDDXJ4ATVA","created_at":"2026-07-05T00:34:24.644179+00:00"},{"alias_kind":"pith_short_16","alias_value":"VYFDDXJ4ATVA6LDK","created_at":"2026-07-05T00:34:24.644179+00:00"},{"alias_kind":"pith_short_8","alias_value":"VYFDDXJ4","created_at":"2026-07-05T00:34:24.644179+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04187","citing_title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK","json":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK.json","graph_json":"https://pith.science/api/pith-number/VYFDDXJ4ATVA6LDKSOZOF2KRYK/graph.json","events_json":"https://pith.science/api/pith-number/VYFDDXJ4ATVA6LDKSOZOF2KRYK/events.json","paper":"https://pith.science/paper/VYFDDXJ4"},"agent_actions":{"view_html":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK","download_json":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK.json","view_paper":"https://pith.science/paper/VYFDDXJ4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2001.06781&json=true","fetch_graph":"https://pith.science/api/pith-number/VYFDDXJ4ATVA6LDKSOZOF2KRYK/graph.json","fetch_events":"https://pith.science/api/pith-number/VYFDDXJ4ATVA6LDKSOZOF2KRYK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/action/storage_attestation","attest_author":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/action/author_attestation","sign_citation":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/action/citation_signature","submit_replication":"https://pith.science/pith/VYFDDXJ4ATVA6LDKSOZOF2KRYK/action/replication_record"}},"created_at":"2026-07-05T00:34:24.644179+00:00","updated_at":"2026-07-05T00:34:24.644179+00:00"}