{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:E3SNX2HMHEZB7CEY74BL2PJYSK","short_pith_number":"pith:E3SNX2HM","schema_version":"1.0","canonical_sha256":"26e4dbe8ec39321f8898ff02bd3d389282d2aab941604df5f435cef5d15b3330","source":{"kind":"arxiv","id":"2006.03647","version":2},"attestation_state":"computed","paper":{"title":"Deployment-Efficient Reinforcement Learning via Model-Based Offline Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Hiroki Furuta, Ofir Nachum, Shixiang Gu, Tatsuya Matsushima, Yutaka Matsuo","submitted_at":"2020-06-05T19:33:19Z","abstract_excerpt":"Most reinforcement learning (RL) algorithms assume online access to the environment, in which one may readily interleave updates to the policy with experience collection using that policy. However, in many real-world applications such as health, education, dialogue agents, and robotics, the cost or potential risk of deploying a new data-collection policy is high, to the point that it can become prohibitive to update the data-collection policy more than a few times during learning. With this view, we propose a novel concept of deployment efficiency, measuring the number of distinct data-collect"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.03647","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-05T19:33:19Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"85607b4288f6cf8cdeb406248068e63b5ae2be353ba9f38354059ad616485d1d","abstract_canon_sha256":"8705e2a15ebf131c57449405f405ea7d5578e1e7d48bb4178cd3372471654053"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:12:48.350221Z","signature_b64":"zcbe9xtNh8f+1vK8AIKTkHrVHyrc7oorMCAAJuvQRSsrhDqrYXKS/SavXpvHiFQhwTU0LiSy1hO5tNWvAg4HDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"26e4dbe8ec39321f8898ff02bd3d389282d2aab941604df5f435cef5d15b3330","last_reissued_at":"2026-07-05T01:12:48.349762Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:12:48.349762Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deployment-Efficient Reinforcement Learning via Model-Based Offline Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Hiroki Furuta, Ofir Nachum, Shixiang Gu, Tatsuya Matsushima, Yutaka Matsuo","submitted_at":"2020-06-05T19:33:19Z","abstract_excerpt":"Most reinforcement learning (RL) algorithms assume online access to the environment, in which one may readily interleave updates to the policy with experience collection using that policy. However, in many real-world applications such as health, education, dialogue agents, and robotics, the cost or potential risk of deploying a new data-collection policy is high, to the point that it can become prohibitive to update the data-collection policy more than a few times during learning. With this view, we propose a novel concept of deployment efficiency, measuring the number of distinct data-collect"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.03647","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.03647/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.03647","created_at":"2026-07-05T01:12:48.349824+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.03647v2","created_at":"2026-07-05T01:12:48.349824+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.03647","created_at":"2026-07-05T01:12:48.349824+00:00"},{"alias_kind":"pith_short_12","alias_value":"E3SNX2HMHEZB","created_at":"2026-07-05T01:12:48.349824+00:00"},{"alias_kind":"pith_short_16","alias_value":"E3SNX2HMHEZB7CEY","created_at":"2026-07-05T01:12:48.349824+00:00"},{"alias_kind":"pith_short_8","alias_value":"E3SNX2HM","created_at":"2026-07-05T01:12:48.349824+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E3SNX2HMHEZB7CEY74BL2PJYSK","json":"https://pith.science/pith/E3SNX2HMHEZB7CEY74BL2PJYSK.json","graph_json":"https://pith.science/api/pith-number/E3SNX2HMHEZB7CEY74BL2PJYSK/graph.json","events_json":"https://pith.science/api/pith-number/E3SNX2HMHEZB7CEY74BL2PJYSK/events.json","paper":"https://pith.science/paper/E3SNX2HM"},"agent_actions":{"view_html":"https://pith.science/pith/E3SNX2HMHEZB7CEY74BL2PJYSK","download_json":"https://pith.science/pith/E3SNX2HMHEZB7CEY74BL2PJYSK.json","view_paper":"https://pith.science/paper/E3SNX2HM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.03647&json=true","fetch_graph":"https://pith.science/api/pith-number/E3SNX2HMHEZB7CEY74BL2PJYSK/graph.json","fetch_events":"https://pith.science/api/pith-number/E3SNX2HMHEZB7CEY74BL2PJYSK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E3SNX2HMHEZB7CEY74BL2PJYSK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E3SNX2HMHEZB7CEY74BL2PJYSK/action/storage_attestation","attest_author":"https://pith.science/pith/E3SNX2HMHEZB7CEY74BL2PJYSK/action/author_attestation","sign_citation":"https://pith.science/pith/E3SNX2HMHEZB7CEY74BL2PJYSK/action/citation_signature","submit_replication":"https://pith.science/pith/E3SNX2HMHEZB7CEY74BL2PJYSK/action/replication_record"}},"created_at":"2026-07-05T01:12:48.349824+00:00","updated_at":"2026-07-05T01:12:48.349824+00:00"}