{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:N3JSUABVGOJCE7CGQIDT66ZOC6","short_pith_number":"pith:N3JSUABV","schema_version":"1.0","canonical_sha256":"6ed32a00353392227c4682073f7b2e17afb7a2bdf38ccb3e0e28949b1eb4cc47","source":{"kind":"arxiv","id":"1907.12439","version":5},"attestation_state":"computed","paper":{"title":"Hindsight Trust Region Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"David Hsu, Hanbo Zhang, Nanning Zheng, Site Bai, Xuguang Lan","submitted_at":"2019-07-29T13:59:42Z","abstract_excerpt":"Reinforcement Learning(RL) with sparse rewards is a major challenge. We propose \\emph{Hindsight Trust Region Policy Optimization}(HTRPO), a new RL algorithm that extends the highly successful TRPO algorithm with \\emph{hindsight} to tackle the challenge of sparse rewards. Hindsight refers to the algorithm's ability to learn from information across goals, including ones not intended for the current task. HTRPO leverages two main ideas. It introduces QKL, a quadratic approximation to the KL divergence constraint on the trust region, leading to reduced variance in KL divergence estimation and impr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1907.12439","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-07-29T13:59:42Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"f92f81221d7b3913b56546bd9321073cd77ab5c9bd6d8b5a8361f63c4df9d462","abstract_canon_sha256":"1d223c2cd107afd1ba55dcd381dcd779ee6ab2f3714251b6b477ef5df1ab56eb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:40:25.495442Z","signature_b64":"vHIfUfbuRaxfyyweC+Jh0K0LtYmjPQphNElfis1/+pD/dOxATc/uotTNemWXTeoFsoGNchZq6zwPVV/gohiNDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6ed32a00353392227c4682073f7b2e17afb7a2bdf38ccb3e0e28949b1eb4cc47","last_reissued_at":"2026-07-05T02:40:25.494941Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:40:25.494941Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hindsight Trust Region Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"David Hsu, Hanbo Zhang, Nanning Zheng, Site Bai, Xuguang Lan","submitted_at":"2019-07-29T13:59:42Z","abstract_excerpt":"Reinforcement Learning(RL) with sparse rewards is a major challenge. We propose \\emph{Hindsight Trust Region Policy Optimization}(HTRPO), a new RL algorithm that extends the highly successful TRPO algorithm with \\emph{hindsight} to tackle the challenge of sparse rewards. Hindsight refers to the algorithm's ability to learn from information across goals, including ones not intended for the current task. HTRPO leverages two main ideas. It introduces QKL, a quadratic approximation to the KL divergence constraint on the trust region, leading to reduced variance in KL divergence estimation and impr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1907.12439","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1907.12439/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1907.12439","created_at":"2026-07-05T02:40:25.495001+00:00"},{"alias_kind":"arxiv_version","alias_value":"1907.12439v5","created_at":"2026-07-05T02:40:25.495001+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1907.12439","created_at":"2026-07-05T02:40:25.495001+00:00"},{"alias_kind":"pith_short_12","alias_value":"N3JSUABVGOJC","created_at":"2026-07-05T02:40:25.495001+00:00"},{"alias_kind":"pith_short_16","alias_value":"N3JSUABVGOJCE7CG","created_at":"2026-07-05T02:40:25.495001+00:00"},{"alias_kind":"pith_short_8","alias_value":"N3JSUABV","created_at":"2026-07-05T02:40:25.495001+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N3JSUABVGOJCE7CGQIDT66ZOC6","json":"https://pith.science/pith/N3JSUABVGOJCE7CGQIDT66ZOC6.json","graph_json":"https://pith.science/api/pith-number/N3JSUABVGOJCE7CGQIDT66ZOC6/graph.json","events_json":"https://pith.science/api/pith-number/N3JSUABVGOJCE7CGQIDT66ZOC6/events.json","paper":"https://pith.science/paper/N3JSUABV"},"agent_actions":{"view_html":"https://pith.science/pith/N3JSUABVGOJCE7CGQIDT66ZOC6","download_json":"https://pith.science/pith/N3JSUABVGOJCE7CGQIDT66ZOC6.json","view_paper":"https://pith.science/paper/N3JSUABV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1907.12439&json=true","fetch_graph":"https://pith.science/api/pith-number/N3JSUABVGOJCE7CGQIDT66ZOC6/graph.json","fetch_events":"https://pith.science/api/pith-number/N3JSUABVGOJCE7CGQIDT66ZOC6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N3JSUABVGOJCE7CGQIDT66ZOC6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N3JSUABVGOJCE7CGQIDT66ZOC6/action/storage_attestation","attest_author":"https://pith.science/pith/N3JSUABVGOJCE7CGQIDT66ZOC6/action/author_attestation","sign_citation":"https://pith.science/pith/N3JSUABVGOJCE7CGQIDT66ZOC6/action/citation_signature","submit_replication":"https://pith.science/pith/N3JSUABVGOJCE7CGQIDT66ZOC6/action/replication_record"}},"created_at":"2026-07-05T02:40:25.495001+00:00","updated_at":"2026-07-05T02:40:25.495001+00:00"}