{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2020:HNZZBQCG63RSWWGFYYJJNNTC7Z","short_pith_number":"pith:HNZZBQCG","canonical_record":{"source":{"id":"2003.02740","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-03-05T16:10:15Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"d17a283058a422a764d91c7ea0546697d3b085c5d9356a447bf0c37b0be98231","abstract_canon_sha256":"131b73b701e0b865038df2bf05692165539be2b9c740a4dd409057fb65c51b67"},"schema_version":"1.0"},"canonical_sha256":"3b7390c046f6e32b58c5c61296b662fe60353f275696f399dd04b09cab44aa10","source":{"kind":"arxiv","id":"2003.02740","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2003.02740","created_at":"2026-07-05T00:45:59Z"},{"alias_kind":"arxiv_version","alias_value":"2003.02740v1","created_at":"2026-07-05T00:45:59Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.02740","created_at":"2026-07-05T00:45:59Z"},{"alias_kind":"pith_short_12","alias_value":"HNZZBQCG63RS","created_at":"2026-07-05T00:45:59Z"},{"alias_kind":"pith_short_16","alias_value":"HNZZBQCG63RSWWGF","created_at":"2026-07-05T00:45:59Z"},{"alias_kind":"pith_short_8","alias_value":"HNZZBQCG","created_at":"2026-07-05T00:45:59Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2020:HNZZBQCG63RSWWGFYYJJNNTC7Z","target":"record","payload":{"canonical_record":{"source":{"id":"2003.02740","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-03-05T16:10:15Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"d17a283058a422a764d91c7ea0546697d3b085c5d9356a447bf0c37b0be98231","abstract_canon_sha256":"131b73b701e0b865038df2bf05692165539be2b9c740a4dd409057fb65c51b67"},"schema_version":"1.0"},"canonical_sha256":"3b7390c046f6e32b58c5c61296b662fe60353f275696f399dd04b09cab44aa10","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:45:59.138425Z","signature_b64":"xXfRtItO5n+4lS9M2jfE4trITHoaEDQ+8DOHQR93zma3Y5EbZrnXYx6zEoH+5r5DuBL97WxYriju9zgffB3lCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3b7390c046f6e32b58c5c61296b662fe60353f275696f399dd04b09cab44aa10","last_reissued_at":"2026-07-05T00:45:59.137916Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:45:59.137916Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2003.02740","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:45:59Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"EuvoFSegNjOvvlBsAEhLQ8dn7KDDciHYOtG8/D4vLW9THN23mCIoVrgThmko26fmUWMwZ+UeMKsC1O/PncEMAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-05T05:42:14.779295Z"},"content_sha256":"41afb11da81ba1995c17a5d88368bd7b7c2fd6224cd25018046fece2b7b3e23e","schema_version":"1.0","event_id":"sha256:41afb11da81ba1995c17a5d88368bd7b7c2fd6224cd25018046fece2b7b3e23e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2020:HNZZBQCG63RSWWGFYYJJNNTC7Z","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Balance Between Efficient and Effective Learning: Dense2Sparse Reward Shaping for Robot Manipulation with Environment Uncertainty","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Bo Song, Chao Zhou, Kun Dong, Lili Zhao, Yongle Luo, Zhiyong Sun","submitted_at":"2020-03-05T16:10:15Z","abstract_excerpt":"Efficient and effective learning is one of the ultimate goals of the deep reinforcement learning (DRL), although the compromise has been made in most of the time, especially for the application of robot manipulations. Learning is always expensive for robot manipulation tasks and the learning effectiveness could be affected by the system uncertainty. In order to solve above challenges, in this study, we proposed a simple but powerful reward shaping method, namely Dense2Sparse. It combines the advantage of fast convergence of dense reward and the noise isolation of the sparse reward, to achieve "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.02740","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2003.02740/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:45:59Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PqI74Jpg9epoWncQ4irDgCa/SR85cUew0Qm1j6lrxHzLyoWSZ2KlFHZKGryxp+l4lkrHQcFl+NJinBBOzoziBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-05T05:42:14.779682Z"},"content_sha256":"b9545b4d71b14a9b95516a9837e989d30798ed082a1c50d276139422fc953c7a","schema_version":"1.0","event_id":"sha256:b9545b4d71b14a9b95516a9837e989d30798ed082a1c50d276139422fc953c7a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/HNZZBQCG63RSWWGFYYJJNNTC7Z/bundle.json","state_url":"https://pith.science/pith/HNZZBQCG63RSWWGFYYJJNNTC7Z/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/HNZZBQCG63RSWWGFYYJJNNTC7Z/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-05T05:42:14Z","links":{"resolver":"https://pith.science/pith/HNZZBQCG63RSWWGFYYJJNNTC7Z","bundle":"https://pith.science/pith/HNZZBQCG63RSWWGFYYJJNNTC7Z/bundle.json","state":"https://pith.science/pith/HNZZBQCG63RSWWGFYYJJNNTC7Z/state.json","well_known_bundle":"https://pith.science/.well-known/pith/HNZZBQCG63RSWWGFYYJJNNTC7Z/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:HNZZBQCG63RSWWGFYYJJNNTC7Z","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"131b73b701e0b865038df2bf05692165539be2b9c740a4dd409057fb65c51b67","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-03-05T16:10:15Z","title_canon_sha256":"d17a283058a422a764d91c7ea0546697d3b085c5d9356a447bf0c37b0be98231"},"schema_version":"1.0","source":{"id":"2003.02740","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2003.02740","created_at":"2026-07-05T00:45:59Z"},{"alias_kind":"arxiv_version","alias_value":"2003.02740v1","created_at":"2026-07-05T00:45:59Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.02740","created_at":"2026-07-05T00:45:59Z"},{"alias_kind":"pith_short_12","alias_value":"HNZZBQCG63RS","created_at":"2026-07-05T00:45:59Z"},{"alias_kind":"pith_short_16","alias_value":"HNZZBQCG63RSWWGF","created_at":"2026-07-05T00:45:59Z"},{"alias_kind":"pith_short_8","alias_value":"HNZZBQCG","created_at":"2026-07-05T00:45:59Z"}],"graph_snapshots":[{"event_id":"sha256:b9545b4d71b14a9b95516a9837e989d30798ed082a1c50d276139422fc953c7a","target":"graph","created_at":"2026-07-05T00:45:59Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2003.02740/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Efficient and effective learning is one of the ultimate goals of the deep reinforcement learning (DRL), although the compromise has been made in most of the time, especially for the application of robot manipulations. Learning is always expensive for robot manipulation tasks and the learning effectiveness could be affected by the system uncertainty. In order to solve above challenges, in this study, we proposed a simple but powerful reward shaping method, namely Dense2Sparse. It combines the advantage of fast convergence of dense reward and the noise isolation of the sparse reward, to achieve ","authors_text":"Bo Song, Chao Zhou, Kun Dong, Lili Zhao, Yongle Luo, Zhiyong Sun","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-03-05T16:10:15Z","title":"Balance Between Efficient and Effective Learning: Dense2Sparse Reward Shaping for Robot Manipulation with Environment Uncertainty"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.02740","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:41afb11da81ba1995c17a5d88368bd7b7c2fd6224cd25018046fece2b7b3e23e","target":"record","created_at":"2026-07-05T00:45:59Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"131b73b701e0b865038df2bf05692165539be2b9c740a4dd409057fb65c51b67","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-03-05T16:10:15Z","title_canon_sha256":"d17a283058a422a764d91c7ea0546697d3b085c5d9356a447bf0c37b0be98231"},"schema_version":"1.0","source":{"id":"2003.02740","kind":"arxiv","version":1}},"canonical_sha256":"3b7390c046f6e32b58c5c61296b662fe60353f275696f399dd04b09cab44aa10","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3b7390c046f6e32b58c5c61296b662fe60353f275696f399dd04b09cab44aa10","first_computed_at":"2026-07-05T00:45:59.137916Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T00:45:59.137916Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"xXfRtItO5n+4lS9M2jfE4trITHoaEDQ+8DOHQR93zma3Y5EbZrnXYx6zEoH+5r5DuBL97WxYriju9zgffB3lCA==","signature_status":"signed_v1","signed_at":"2026-07-05T00:45:59.138425Z","signed_message":"canonical_sha256_bytes"},"source_id":"2003.02740","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:41afb11da81ba1995c17a5d88368bd7b7c2fd6224cd25018046fece2b7b3e23e","sha256:b9545b4d71b14a9b95516a9837e989d30798ed082a1c50d276139422fc953c7a"],"state_sha256":"db2550c00db869e4c522231d97e533ccec702e2999e783841ae5f2ff6d6dc4d5"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"96UFpedmZJ6XAODnPAY+s6kQwdiW+wGkkJsxeP4pcMuZPTp6AOMpjrUiPQLBeJRrfSRejp6ouxPgecA7KYvgDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-05T05:42:14.781568Z","bundle_sha256":"d31f7ce033cbc28f7f87b042176c5120bfdc8e8c7542f5ea71cddef077ee9f71"}}