{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:V26D26QTWFDQJQSBQWQ2X4PRF6","short_pith_number":"pith:V26D26QT","canonical_record":{"source":{"id":"2202.04628","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-09T18:45:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"066b9316b360db0e36822fb81d634db4e2e5bf41e96e8a9ffb7cb85a91270d85","abstract_canon_sha256":"4c36e1dce4c97d06efc37b4b161b9f0593d51197bd21ebfd086c2775b2f45a26"},"schema_version":"1.0"},"canonical_sha256":"aebc3d7a13b14704c24185a1abf1f12f8296b45a16ecac6c4a585a2dc4b15851","source":{"kind":"arxiv","id":"2202.04628","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2202.04628","created_at":"2026-07-05T03:56:34Z"},{"alias_kind":"arxiv_version","alias_value":"2202.04628v2","created_at":"2026-07-05T03:56:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.04628","created_at":"2026-07-05T03:56:34Z"},{"alias_kind":"pith_short_12","alias_value":"V26D26QTWFDQ","created_at":"2026-07-05T03:56:34Z"},{"alias_kind":"pith_short_16","alias_value":"V26D26QTWFDQJQSB","created_at":"2026-07-05T03:56:34Z"},{"alias_kind":"pith_short_8","alias_value":"V26D26QT","created_at":"2026-07-05T03:56:34Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:V26D26QTWFDQJQSBQWQ2X4PRF6","target":"record","payload":{"canonical_record":{"source":{"id":"2202.04628","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-09T18:45:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"066b9316b360db0e36822fb81d634db4e2e5bf41e96e8a9ffb7cb85a91270d85","abstract_canon_sha256":"4c36e1dce4c97d06efc37b4b161b9f0593d51197bd21ebfd086c2775b2f45a26"},"schema_version":"1.0"},"canonical_sha256":"aebc3d7a13b14704c24185a1abf1f12f8296b45a16ecac6c4a585a2dc4b15851","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:56:34.649958Z","signature_b64":"IEe0t+n7MHUH3eDg+eXycaFwN6iJBteEsGLm2hKaABrTf9KzJjycgtqhFVV1dA/u2D4m6opvbP/Rpkjyz1LoBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aebc3d7a13b14704c24185a1abf1f12f8296b45a16ecac6c4a585a2dc4b15851","last_reissued_at":"2026-07-05T03:56:34.649477Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:56:34.649477Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2202.04628","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:56:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"e7nVlo65MEJHYP2VlMKo+2t2vhqcbAI6M9w6FElqGRbwwypPhjC9HF4UiJe1K2P2/3ENBT96UgnVJmvHUEr8Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T18:28:39.055158Z"},"content_sha256":"d9f34b1df74170345a1437e3d6290f0c238ed0089b2b7bd3529324c9e77a4b9a","schema_version":"1.0","event_id":"sha256:d9f34b1df74170345a1437e3d6290f0c238ed0089b2b7bd3529324c9e77a4b9a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:V26D26QTWFDQJQSBQWQ2X4PRF6","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reinforcement Learning with Sparse Rewards using Guidance from Offline Demonstration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Akshay Sarvesh, Desik Rengarajan, Dileep Kalathil, Gargi Vaidya, Srinivas Shakkottai","submitted_at":"2022-02-09T18:45:40Z","abstract_excerpt":"A major challenge in real-world reinforcement learning (RL) is the sparsity of reward feedback. Often, what is available is an intuitive but sparse reward function that only indicates whether the task is completed partially or fully. However, the lack of carefully designed, fine grain feedback implies that most existing RL algorithms fail to learn an acceptable policy in a reasonable time frame. This is because of the large number of exploration actions that the policy has to perform before it gets any useful feedback that it can learn from. In this work, we address this challenging problem by"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.04628","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.04628/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:56:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PBi9+nGUnGtT791GtP7Z4DncYkip16doNatlbVHd+MgONFw7zio8i+O/cAhyk4oD4dlOyyRb7exEcsUCbQf0Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T18:28:39.055712Z"},"content_sha256":"859a1d6ef156ae0f1089c17493030da273dc5355c8f2348395cb8bea0c051071","schema_version":"1.0","event_id":"sha256:859a1d6ef156ae0f1089c17493030da273dc5355c8f2348395cb8bea0c051071"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/V26D26QTWFDQJQSBQWQ2X4PRF6/bundle.json","state_url":"https://pith.science/pith/V26D26QTWFDQJQSBQWQ2X4PRF6/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/V26D26QTWFDQJQSBQWQ2X4PRF6/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T18:28:39Z","links":{"resolver":"https://pith.science/pith/V26D26QTWFDQJQSBQWQ2X4PRF6","bundle":"https://pith.science/pith/V26D26QTWFDQJQSBQWQ2X4PRF6/bundle.json","state":"https://pith.science/pith/V26D26QTWFDQJQSBQWQ2X4PRF6/state.json","well_known_bundle":"https://pith.science/.well-known/pith/V26D26QTWFDQJQSBQWQ2X4PRF6/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:V26D26QTWFDQJQSBQWQ2X4PRF6","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4c36e1dce4c97d06efc37b4b161b9f0593d51197bd21ebfd086c2775b2f45a26","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-09T18:45:40Z","title_canon_sha256":"066b9316b360db0e36822fb81d634db4e2e5bf41e96e8a9ffb7cb85a91270d85"},"schema_version":"1.0","source":{"id":"2202.04628","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2202.04628","created_at":"2026-07-05T03:56:34Z"},{"alias_kind":"arxiv_version","alias_value":"2202.04628v2","created_at":"2026-07-05T03:56:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.04628","created_at":"2026-07-05T03:56:34Z"},{"alias_kind":"pith_short_12","alias_value":"V26D26QTWFDQ","created_at":"2026-07-05T03:56:34Z"},{"alias_kind":"pith_short_16","alias_value":"V26D26QTWFDQJQSB","created_at":"2026-07-05T03:56:34Z"},{"alias_kind":"pith_short_8","alias_value":"V26D26QT","created_at":"2026-07-05T03:56:34Z"}],"graph_snapshots":[{"event_id":"sha256:859a1d6ef156ae0f1089c17493030da273dc5355c8f2348395cb8bea0c051071","target":"graph","created_at":"2026-07-05T03:56:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2202.04628/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"A major challenge in real-world reinforcement learning (RL) is the sparsity of reward feedback. Often, what is available is an intuitive but sparse reward function that only indicates whether the task is completed partially or fully. However, the lack of carefully designed, fine grain feedback implies that most existing RL algorithms fail to learn an acceptable policy in a reasonable time frame. This is because of the large number of exploration actions that the policy has to perform before it gets any useful feedback that it can learn from. In this work, we address this challenging problem by","authors_text":"Akshay Sarvesh, Desik Rengarajan, Dileep Kalathil, Gargi Vaidya, Srinivas Shakkottai","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-09T18:45:40Z","title":"Reinforcement Learning with Sparse Rewards using Guidance from Offline Demonstration"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.04628","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d9f34b1df74170345a1437e3d6290f0c238ed0089b2b7bd3529324c9e77a4b9a","target":"record","created_at":"2026-07-05T03:56:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4c36e1dce4c97d06efc37b4b161b9f0593d51197bd21ebfd086c2775b2f45a26","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-09T18:45:40Z","title_canon_sha256":"066b9316b360db0e36822fb81d634db4e2e5bf41e96e8a9ffb7cb85a91270d85"},"schema_version":"1.0","source":{"id":"2202.04628","kind":"arxiv","version":2}},"canonical_sha256":"aebc3d7a13b14704c24185a1abf1f12f8296b45a16ecac6c4a585a2dc4b15851","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"aebc3d7a13b14704c24185a1abf1f12f8296b45a16ecac6c4a585a2dc4b15851","first_computed_at":"2026-07-05T03:56:34.649477Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:56:34.649477Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"IEe0t+n7MHUH3eDg+eXycaFwN6iJBteEsGLm2hKaABrTf9KzJjycgtqhFVV1dA/u2D4m6opvbP/Rpkjyz1LoBw==","signature_status":"signed_v1","signed_at":"2026-07-05T03:56:34.649958Z","signed_message":"canonical_sha256_bytes"},"source_id":"2202.04628","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d9f34b1df74170345a1437e3d6290f0c238ed0089b2b7bd3529324c9e77a4b9a","sha256:859a1d6ef156ae0f1089c17493030da273dc5355c8f2348395cb8bea0c051071"],"state_sha256":"90a7795124cf07229e40d7aa7da13240f12d16ea719e28268c3bcf8d4b0dc3d8"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"tADS2PUN7ic1xOwmVIzgujKNKUNmxbX+f2ARHpSDl18A7mpVNDd/jenwYDZ128er1at3CvcDC1Dg2tAYi6jZBQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T18:28:39.060710Z","bundle_sha256":"e8719afc99da08e519ced37882c414718f1e58c1ea109dd6d3872c5b3b54647e"}}