{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MVUATXYVZSUNVXNPVDTEYJCS5M","short_pith_number":"pith:MVUATXYV","schema_version":"1.0","canonical_sha256":"656809df15cca8daddafa8e64c2452eb2231bd041e20f63c7260a376dcda1b53","source":{"kind":"arxiv","id":"2505.22442","version":2},"attestation_state":"computed","paper":{"title":"SOReL and TOReL: Two Methods for Fully Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Clarisse Wibault, Jakob N. Foerster, Johannes Forkel, Mattie Fellows, Michael A. Osborne, Uljad Berdica","submitted_at":"2025-05-28T15:07:24Z","abstract_excerpt":"Sample efficiency remains a major obstacle for real world adoption of reinforcement learning (RL): success has been limited to settings where simulators provide access to essentially unlimited environment interactions, which in reality are typically costly or dangerous to obtain. Offline RL in principle offers a solution by exploiting offline data to learn a near-optimal policy before deployment. In practice, however, current offline RL methods rely on extensive online interactions for hyperparameter tuning, and have no reliable bound on their initial online performance. To address these two i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22442","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-28T15:07:24Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"626584707144a66efd6cfda9cea45ee4584b6343233bcafe94c91058f561be56","abstract_canon_sha256":"c25da1cc4d6e9d2452672539c6a868d4bf30d5ae9bba4aef0494ed9eecb6058e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:34.165161Z","signature_b64":"LDB9F7LQEhLfCkseJ/ayK3XHc27W5CNhekkCGATzZKXjFFyvCw+71xkO5Obkf7Oy4/5p91BLlanPHfcffxD+BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"656809df15cca8daddafa8e64c2452eb2231bd041e20f63c7260a376dcda1b53","last_reissued_at":"2026-07-05T11:12:34.164622Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:34.164622Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SOReL and TOReL: Two Methods for Fully Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Clarisse Wibault, Jakob N. Foerster, Johannes Forkel, Mattie Fellows, Michael A. Osborne, Uljad Berdica","submitted_at":"2025-05-28T15:07:24Z","abstract_excerpt":"Sample efficiency remains a major obstacle for real world adoption of reinforcement learning (RL): success has been limited to settings where simulators provide access to essentially unlimited environment interactions, which in reality are typically costly or dangerous to obtain. Offline RL in principle offers a solution by exploiting offline data to learn a near-optimal policy before deployment. In practice, however, current offline RL methods rely on extensive online interactions for hyperparameter tuning, and have no reliable bound on their initial online performance. To address these two i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22442","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22442/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22442","created_at":"2026-07-05T11:12:34.164694+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22442v2","created_at":"2026-07-05T11:12:34.164694+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22442","created_at":"2026-07-05T11:12:34.164694+00:00"},{"alias_kind":"pith_short_12","alias_value":"MVUATXYVZSUN","created_at":"2026-07-05T11:12:34.164694+00:00"},{"alias_kind":"pith_short_16","alias_value":"MVUATXYVZSUNVXNP","created_at":"2026-07-05T11:12:34.164694+00:00"},{"alias_kind":"pith_short_8","alias_value":"MVUATXYV","created_at":"2026-07-05T11:12:34.164694+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2512.04341","citing_title":"Long-Horizon Model-Based Offline Reinforcement Learning Without Explicit Conservatism","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MVUATXYVZSUNVXNPVDTEYJCS5M","json":"https://pith.science/pith/MVUATXYVZSUNVXNPVDTEYJCS5M.json","graph_json":"https://pith.science/api/pith-number/MVUATXYVZSUNVXNPVDTEYJCS5M/graph.json","events_json":"https://pith.science/api/pith-number/MVUATXYVZSUNVXNPVDTEYJCS5M/events.json","paper":"https://pith.science/paper/MVUATXYV"},"agent_actions":{"view_html":"https://pith.science/pith/MVUATXYVZSUNVXNPVDTEYJCS5M","download_json":"https://pith.science/pith/MVUATXYVZSUNVXNPVDTEYJCS5M.json","view_paper":"https://pith.science/paper/MVUATXYV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22442&json=true","fetch_graph":"https://pith.science/api/pith-number/MVUATXYVZSUNVXNPVDTEYJCS5M/graph.json","fetch_events":"https://pith.science/api/pith-number/MVUATXYVZSUNVXNPVDTEYJCS5M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MVUATXYVZSUNVXNPVDTEYJCS5M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MVUATXYVZSUNVXNPVDTEYJCS5M/action/storage_attestation","attest_author":"https://pith.science/pith/MVUATXYVZSUNVXNPVDTEYJCS5M/action/author_attestation","sign_citation":"https://pith.science/pith/MVUATXYVZSUNVXNPVDTEYJCS5M/action/citation_signature","submit_replication":"https://pith.science/pith/MVUATXYVZSUNVXNPVDTEYJCS5M/action/replication_record"}},"created_at":"2026-07-05T11:12:34.164694+00:00","updated_at":"2026-07-05T11:12:34.164694+00:00"}