{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:DCA2OZD7YLTT2SCDZWQDJ7PTH5","short_pith_number":"pith:DCA2OZD7","schema_version":"1.0","canonical_sha256":"1881a7647fc2e73d4843cda034fdf33f4ab3691c35e36a04b74e360a27e36253","source":{"kind":"arxiv","id":"2212.08232","version":1},"attestation_state":"computed","paper":{"title":"Offline Robot Reinforcement Learning with Uncertainty-Guided Human Expert Sampling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Ashish Kumar, Ilya Kuzovkin","submitted_at":"2022-12-16T01:41:59Z","abstract_excerpt":"Recent advances in batch (offline) reinforcement learning have shown promising results in learning from available offline data and proved offline reinforcement learning to be an essential toolkit in learning control policies in a model-free setting. An offline reinforcement learning algorithm applied to a dataset collected by a suboptimal non-learning-based algorithm can result in a policy that outperforms the behavior agent used to collect the data. Such a scenario is frequent in robotics, where existing automation is collecting operational data. Although offline learning techniques can learn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.08232","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-12-16T01:41:59Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"6acbaa3726b13984f384484d207a0a6e1b8e55e7f7964e3014f28380c11e03d0","abstract_canon_sha256":"1688e01dbb944793309c40cc91c987941b406bda5b98f8e2c913d8bdfb5773be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:25:54.193113Z","signature_b64":"uNNuaD7HIceQK1UbpYzWZws+2V+iPdj57LOck6Itkp3Co+ALPi9HoYM6eEYPCBJow9f9DYMri6rDL89zlM4aCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1881a7647fc2e73d4843cda034fdf33f4ab3691c35e36a04b74e360a27e36253","last_reissued_at":"2026-07-05T05:25:54.192717Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:25:54.192717Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Offline Robot Reinforcement Learning with Uncertainty-Guided Human Expert Sampling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Ashish Kumar, Ilya Kuzovkin","submitted_at":"2022-12-16T01:41:59Z","abstract_excerpt":"Recent advances in batch (offline) reinforcement learning have shown promising results in learning from available offline data and proved offline reinforcement learning to be an essential toolkit in learning control policies in a model-free setting. An offline reinforcement learning algorithm applied to a dataset collected by a suboptimal non-learning-based algorithm can result in a policy that outperforms the behavior agent used to collect the data. Such a scenario is frequent in robotics, where existing automation is collecting operational data. Although offline learning techniques can learn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.08232","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.08232/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.08232","created_at":"2026-07-05T05:25:54.192780+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.08232v1","created_at":"2026-07-05T05:25:54.192780+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.08232","created_at":"2026-07-05T05:25:54.192780+00:00"},{"alias_kind":"pith_short_12","alias_value":"DCA2OZD7YLTT","created_at":"2026-07-05T05:25:54.192780+00:00"},{"alias_kind":"pith_short_16","alias_value":"DCA2OZD7YLTT2SCD","created_at":"2026-07-05T05:25:54.192780+00:00"},{"alias_kind":"pith_short_8","alias_value":"DCA2OZD7","created_at":"2026-07-05T05:25:54.192780+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.07302","citing_title":"Application of LLMs to Multi-Robot Path Planning and Task Allocation","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DCA2OZD7YLTT2SCDZWQDJ7PTH5","json":"https://pith.science/pith/DCA2OZD7YLTT2SCDZWQDJ7PTH5.json","graph_json":"https://pith.science/api/pith-number/DCA2OZD7YLTT2SCDZWQDJ7PTH5/graph.json","events_json":"https://pith.science/api/pith-number/DCA2OZD7YLTT2SCDZWQDJ7PTH5/events.json","paper":"https://pith.science/paper/DCA2OZD7"},"agent_actions":{"view_html":"https://pith.science/pith/DCA2OZD7YLTT2SCDZWQDJ7PTH5","download_json":"https://pith.science/pith/DCA2OZD7YLTT2SCDZWQDJ7PTH5.json","view_paper":"https://pith.science/paper/DCA2OZD7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.08232&json=true","fetch_graph":"https://pith.science/api/pith-number/DCA2OZD7YLTT2SCDZWQDJ7PTH5/graph.json","fetch_events":"https://pith.science/api/pith-number/DCA2OZD7YLTT2SCDZWQDJ7PTH5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DCA2OZD7YLTT2SCDZWQDJ7PTH5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DCA2OZD7YLTT2SCDZWQDJ7PTH5/action/storage_attestation","attest_author":"https://pith.science/pith/DCA2OZD7YLTT2SCDZWQDJ7PTH5/action/author_attestation","sign_citation":"https://pith.science/pith/DCA2OZD7YLTT2SCDZWQDJ7PTH5/action/citation_signature","submit_replication":"https://pith.science/pith/DCA2OZD7YLTT2SCDZWQDJ7PTH5/action/replication_record"}},"created_at":"2026-07-05T05:25:54.192780+00:00","updated_at":"2026-07-05T05:25:54.192780+00:00"}