{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EISQJMDF2TA7N65BWJRL734V2N","short_pith_number":"pith:EISQJMDF","schema_version":"1.0","canonical_sha256":"222504b065d4c1f6fba1b262bfef95d379992f225b8617a1d92dcbb781c0ed0a","source":{"kind":"arxiv","id":"2508.06108","version":1},"attestation_state":"computed","paper":{"title":"GCHR : Goal-Conditioned Hindsight Regularization for Sample-Efficient Reinforcement Learning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Donglin Wang, Joni Pajarinen, Kaiqiang Ke, Shentao Yang, Wenyan Yang, Xing Lei, Xuetao Zhang","submitted_at":"2025-08-08T08:12:14Z","abstract_excerpt":"Goal-conditioned reinforcement learning (GCRL) with sparse rewards remains a fundamental challenge in reinforcement learning. While hindsight experience replay (HER) has shown promise by relabeling collected trajectories with achieved goals, we argue that trajectory relabeling alone does not fully exploit the available experiences in off-policy GCRL methods, resulting in limited sample efficiency. In this paper, we propose Hindsight Goal-conditioned Regularization (HGR), a technique that generates action regularization priors based on hindsight goals. When combined with hindsight self-imitatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.06108","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-08T08:12:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0ec96c6b718faf0789d60b4f7019593d0f31504d95bbd9f1e1cd23ba926f3fd8","abstract_canon_sha256":"bd82b389205b2f620430fbc1bfe88953e96022621c5c1af37b43f813b5b5d71a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:50:49.953269Z","signature_b64":"d2OWZD4BaIhrr5zcRB4SpCU2Cq7J98UIjDdBHCJeU7G6uxCrgnVsR8uQc46B6Lc53KrvFRCPTS0a7oP9TGONCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"222504b065d4c1f6fba1b262bfef95d379992f225b8617a1d92dcbb781c0ed0a","last_reissued_at":"2026-07-05T11:50:49.952753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:50:49.952753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GCHR : Goal-Conditioned Hindsight Regularization for Sample-Efficient Reinforcement Learning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Donglin Wang, Joni Pajarinen, Kaiqiang Ke, Shentao Yang, Wenyan Yang, Xing Lei, Xuetao Zhang","submitted_at":"2025-08-08T08:12:14Z","abstract_excerpt":"Goal-conditioned reinforcement learning (GCRL) with sparse rewards remains a fundamental challenge in reinforcement learning. While hindsight experience replay (HER) has shown promise by relabeling collected trajectories with achieved goals, we argue that trajectory relabeling alone does not fully exploit the available experiences in off-policy GCRL methods, resulting in limited sample efficiency. In this paper, we propose Hindsight Goal-conditioned Regularization (HGR), a technique that generates action regularization priors based on hindsight goals. When combined with hindsight self-imitatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.06108","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.06108/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.06108","created_at":"2026-07-05T11:50:49.952803+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.06108v1","created_at":"2026-07-05T11:50:49.952803+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.06108","created_at":"2026-07-05T11:50:49.952803+00:00"},{"alias_kind":"pith_short_12","alias_value":"EISQJMDF2TA7","created_at":"2026-07-05T11:50:49.952803+00:00"},{"alias_kind":"pith_short_16","alias_value":"EISQJMDF2TA7N65B","created_at":"2026-07-05T11:50:49.952803+00:00"},{"alias_kind":"pith_short_8","alias_value":"EISQJMDF","created_at":"2026-07-05T11:50:49.952803+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07855","citing_title":"NFTR: From Provable Mode-Averaging to Geodesic Subgoal Selection in Offline Goal-Conditioned RL","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EISQJMDF2TA7N65BWJRL734V2N","json":"https://pith.science/pith/EISQJMDF2TA7N65BWJRL734V2N.json","graph_json":"https://pith.science/api/pith-number/EISQJMDF2TA7N65BWJRL734V2N/graph.json","events_json":"https://pith.science/api/pith-number/EISQJMDF2TA7N65BWJRL734V2N/events.json","paper":"https://pith.science/paper/EISQJMDF"},"agent_actions":{"view_html":"https://pith.science/pith/EISQJMDF2TA7N65BWJRL734V2N","download_json":"https://pith.science/pith/EISQJMDF2TA7N65BWJRL734V2N.json","view_paper":"https://pith.science/paper/EISQJMDF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.06108&json=true","fetch_graph":"https://pith.science/api/pith-number/EISQJMDF2TA7N65BWJRL734V2N/graph.json","fetch_events":"https://pith.science/api/pith-number/EISQJMDF2TA7N65BWJRL734V2N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EISQJMDF2TA7N65BWJRL734V2N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EISQJMDF2TA7N65BWJRL734V2N/action/storage_attestation","attest_author":"https://pith.science/pith/EISQJMDF2TA7N65BWJRL734V2N/action/author_attestation","sign_citation":"https://pith.science/pith/EISQJMDF2TA7N65BWJRL734V2N/action/citation_signature","submit_replication":"https://pith.science/pith/EISQJMDF2TA7N65BWJRL734V2N/action/replication_record"}},"created_at":"2026-07-05T11:50:49.952803+00:00","updated_at":"2026-07-05T11:50:49.952803+00:00"}