{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3LLEX55CXQLVWLWWUG7HWVQ3DR","short_pith_number":"pith:3LLEX55C","schema_version":"1.0","canonical_sha256":"dad64bf7a2bc175b2ed6a1be7b561b1c6dca2a742b40de879aee5e1bd223b5a1","source":{"kind":"arxiv","id":"2412.15525","version":1},"attestation_state":"computed","paper":{"title":"Generalized Back-Stepping Experience Replay in Sparse-Reward Environments","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Guwen Lyu, Masahiro Sato","submitted_at":"2024-12-20T03:31:23Z","abstract_excerpt":"Back-stepping experience replay (BER) is a reinforcement learning technique that can accelerate learning efficiency in reversible environments. BER trains an agent with generated back-stepping transitions of collected experiences and normal forward transitions. However, the original algorithm is designed for a dense-reward environment that does not require complex exploration, limiting the BER technique to demonstrate its full potential. Herein, we propose an enhanced version of BER called Generalized BER (GBER), which extends the original algorithm to sparse-reward environments, particularly "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15525","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-20T03:31:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e1285f6e38d6761ad49e75901514af8d97d8f2e1147aecf959c9fe37626ed8c6","abstract_canon_sha256":"8bf4be17ec48214d28e79db32c09b7ee821dd4822087b18bcdee612968a13e3a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:24.635838Z","signature_b64":"7cIEUNPv3NItUEA8Jeug1pmuxOkaVl6wpMetebwSftghtG6sO8mb07fvNjGKC13+ZApw5FDu5/ACZYeTXl3GDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dad64bf7a2bc175b2ed6a1be7b561b1c6dca2a742b40de879aee5e1bd223b5a1","last_reissued_at":"2026-07-05T09:52:24.635408Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:24.635408Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalized Back-Stepping Experience Replay in Sparse-Reward Environments","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Guwen Lyu, Masahiro Sato","submitted_at":"2024-12-20T03:31:23Z","abstract_excerpt":"Back-stepping experience replay (BER) is a reinforcement learning technique that can accelerate learning efficiency in reversible environments. BER trains an agent with generated back-stepping transitions of collected experiences and normal forward transitions. However, the original algorithm is designed for a dense-reward environment that does not require complex exploration, limiting the BER technique to demonstrate its full potential. Herein, we propose an enhanced version of BER called Generalized BER (GBER), which extends the original algorithm to sparse-reward environments, particularly "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15525","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15525/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15525","created_at":"2026-07-05T09:52:24.635477+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15525v1","created_at":"2026-07-05T09:52:24.635477+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15525","created_at":"2026-07-05T09:52:24.635477+00:00"},{"alias_kind":"pith_short_12","alias_value":"3LLEX55CXQLV","created_at":"2026-07-05T09:52:24.635477+00:00"},{"alias_kind":"pith_short_16","alias_value":"3LLEX55CXQLVWLWW","created_at":"2026-07-05T09:52:24.635477+00:00"},{"alias_kind":"pith_short_8","alias_value":"3LLEX55C","created_at":"2026-07-05T09:52:24.635477+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3LLEX55CXQLVWLWWUG7HWVQ3DR","json":"https://pith.science/pith/3LLEX55CXQLVWLWWUG7HWVQ3DR.json","graph_json":"https://pith.science/api/pith-number/3LLEX55CXQLVWLWWUG7HWVQ3DR/graph.json","events_json":"https://pith.science/api/pith-number/3LLEX55CXQLVWLWWUG7HWVQ3DR/events.json","paper":"https://pith.science/paper/3LLEX55C"},"agent_actions":{"view_html":"https://pith.science/pith/3LLEX55CXQLVWLWWUG7HWVQ3DR","download_json":"https://pith.science/pith/3LLEX55CXQLVWLWWUG7HWVQ3DR.json","view_paper":"https://pith.science/paper/3LLEX55C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15525&json=true","fetch_graph":"https://pith.science/api/pith-number/3LLEX55CXQLVWLWWUG7HWVQ3DR/graph.json","fetch_events":"https://pith.science/api/pith-number/3LLEX55CXQLVWLWWUG7HWVQ3DR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3LLEX55CXQLVWLWWUG7HWVQ3DR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3LLEX55CXQLVWLWWUG7HWVQ3DR/action/storage_attestation","attest_author":"https://pith.science/pith/3LLEX55CXQLVWLWWUG7HWVQ3DR/action/author_attestation","sign_citation":"https://pith.science/pith/3LLEX55CXQLVWLWWUG7HWVQ3DR/action/citation_signature","submit_replication":"https://pith.science/pith/3LLEX55CXQLVWLWWUG7HWVQ3DR/action/replication_record"}},"created_at":"2026-07-05T09:52:24.635477+00:00","updated_at":"2026-07-05T09:52:24.635477+00:00"}