{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZGHOB3V2OA4VPGBG7OKB6GWNDA","short_pith_number":"pith:ZGHOB3V2","schema_version":"1.0","canonical_sha256":"c98ee0eeba7039579826fb941f1acd180be7a9b13fb918edcc24ba992893acd0","source":{"kind":"arxiv","id":"2407.12448","version":2},"attestation_state":"computed","paper":{"title":"Energy-Guided Diffusion Sampling for Offline-to-Online Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ruifeng Chen, Shengyi Jiang, Tian-Shuo Liu, Xinwei Chen, Xu-Hui Liu, Yang Yu, Zhilong Zhang","submitted_at":"2024-07-17T09:56:51Z","abstract_excerpt":"Combining offline and online reinforcement learning (RL) techniques is indeed crucial for achieving efficient and safe learning where data acquisition is expensive. Existing methods replay offline data directly in the online phase, resulting in a significant challenge of data distribution shift and subsequently causing inefficiency in online fine-tuning. To address this issue, we introduce an innovative approach, \\textbf{E}nergy-guided \\textbf{DI}ffusion \\textbf{S}ampling (EDIS), which utilizes a diffusion model to extract prior knowledge from the offline dataset and employs energy functions t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.12448","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-17T09:56:51Z","cross_cats_sorted":[],"title_canon_sha256":"13f4e2cab4f44f7b39f66765e95f7ea846d39ad2691b9b43857dc59d51749170","abstract_canon_sha256":"b2f18d47806eeb923871fa0f7f2fc78292ab76752b42ffabb533a7386cc27090"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:02:41.510547Z","signature_b64":"n8rNniv5gaxvIfCtTbw68Aub08xsSehAPNchr1ouTgHxZgDTjph2SVIvX6MZtHjVpOEkLmZbNSMCGoP9kRQqDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c98ee0eeba7039579826fb941f1acd180be7a9b13fb918edcc24ba992893acd0","last_reissued_at":"2026-07-05T09:02:41.509987Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:02:41.509987Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Energy-Guided Diffusion Sampling for Offline-to-Online Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ruifeng Chen, Shengyi Jiang, Tian-Shuo Liu, Xinwei Chen, Xu-Hui Liu, Yang Yu, Zhilong Zhang","submitted_at":"2024-07-17T09:56:51Z","abstract_excerpt":"Combining offline and online reinforcement learning (RL) techniques is indeed crucial for achieving efficient and safe learning where data acquisition is expensive. Existing methods replay offline data directly in the online phase, resulting in a significant challenge of data distribution shift and subsequently causing inefficiency in online fine-tuning. To address this issue, we introduce an innovative approach, \\textbf{E}nergy-guided \\textbf{DI}ffusion \\textbf{S}ampling (EDIS), which utilizes a diffusion model to extract prior knowledge from the offline dataset and employs energy functions t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12448","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.12448/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.12448","created_at":"2026-07-05T09:02:41.510058+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.12448v2","created_at":"2026-07-05T09:02:41.510058+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12448","created_at":"2026-07-05T09:02:41.510058+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZGHOB3V2OA4V","created_at":"2026-07-05T09:02:41.510058+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZGHOB3V2OA4VPGBG","created_at":"2026-07-05T09:02:41.510058+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZGHOB3V2","created_at":"2026-07-05T09:02:41.510058+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.03828","citing_title":"From Static Constraints to Dynamic Adaptation: Sample-Level Constraint Relaxation for Offline-to-Online Reinforcement Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2511.03828","citing_title":"From Static Constraints to Dynamic Adaptation: Sample-Level Constraint Relaxation for Offline-to-Online Reinforcement Learning","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZGHOB3V2OA4VPGBG7OKB6GWNDA","json":"https://pith.science/pith/ZGHOB3V2OA4VPGBG7OKB6GWNDA.json","graph_json":"https://pith.science/api/pith-number/ZGHOB3V2OA4VPGBG7OKB6GWNDA/graph.json","events_json":"https://pith.science/api/pith-number/ZGHOB3V2OA4VPGBG7OKB6GWNDA/events.json","paper":"https://pith.science/paper/ZGHOB3V2"},"agent_actions":{"view_html":"https://pith.science/pith/ZGHOB3V2OA4VPGBG7OKB6GWNDA","download_json":"https://pith.science/pith/ZGHOB3V2OA4VPGBG7OKB6GWNDA.json","view_paper":"https://pith.science/paper/ZGHOB3V2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.12448&json=true","fetch_graph":"https://pith.science/api/pith-number/ZGHOB3V2OA4VPGBG7OKB6GWNDA/graph.json","fetch_events":"https://pith.science/api/pith-number/ZGHOB3V2OA4VPGBG7OKB6GWNDA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZGHOB3V2OA4VPGBG7OKB6GWNDA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZGHOB3V2OA4VPGBG7OKB6GWNDA/action/storage_attestation","attest_author":"https://pith.science/pith/ZGHOB3V2OA4VPGBG7OKB6GWNDA/action/author_attestation","sign_citation":"https://pith.science/pith/ZGHOB3V2OA4VPGBG7OKB6GWNDA/action/citation_signature","submit_replication":"https://pith.science/pith/ZGHOB3V2OA4VPGBG7OKB6GWNDA/action/replication_record"}},"created_at":"2026-07-05T09:02:41.510058+00:00","updated_at":"2026-07-05T09:02:41.510058+00:00"}