{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:W2KDK4SBOAGWRJZEMMI6LQA2DZ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"a15ef63c035f905e85b44bc6c7bfa00a73baf9f7ff8778042bb12806c06f6f03","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2019-08-15T13:56:12Z","title_canon_sha256":"66e730eed64a15087510604eb5404d8bdafd92e603f5d56660e20ca00b180660"},"schema_version":"1.0","source":{"id":"1908.05546","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1908.05546","created_at":"2026-07-04T23:56:53Z"},{"alias_kind":"arxiv_version","alias_value":"1908.05546v1","created_at":"2026-07-04T23:56:53Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.05546","created_at":"2026-07-04T23:56:53Z"},{"alias_kind":"pith_short_12","alias_value":"W2KDK4SBOAGW","created_at":"2026-07-04T23:56:53Z"},{"alias_kind":"pith_short_16","alias_value":"W2KDK4SBOAGWRJZE","created_at":"2026-07-04T23:56:53Z"},{"alias_kind":"pith_short_8","alias_value":"W2KDK4SB","created_at":"2026-07-04T23:56:53Z"}],"graph_snapshots":[{"event_id":"sha256:6e014d3529d41a362c147ef934d5a4695cdde917a370077c6128aaea6e521f39","target":"graph","created_at":"2026-07-04T23:56:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1908.05546/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Deep reinforcement learning has proven to be a great success in allowing agents to learn complex tasks. However, its application to actual robots can be prohibitively expensive. Furthermore, the unpredictability of human behavior in human-robot interaction tasks can hinder convergence to a good policy. In this paper, we present an architecture that allows agents to learn models of stochastic environments and use them to accelerate learning. We descirbe how an environment model can be learned online and used to generate synthetic transitions, as well as how an agent can leverage these synthetic","authors_text":"Angelo Cangelosi, Massimiliano Patacchiola, Mohammad Thabet","cross_cats":["cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2019-08-15T13:56:12Z","title":"Sample-efficient Deep Reinforcement Learning with Imaginary Rollouts for Human-Robot Interaction"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.05546","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d212a6cda86188a9545c17cbdf48e4fbbea80de3f9a781edc700ce5e62654d7f","target":"record","created_at":"2026-07-04T23:56:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"a15ef63c035f905e85b44bc6c7bfa00a73baf9f7ff8778042bb12806c06f6f03","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2019-08-15T13:56:12Z","title_canon_sha256":"66e730eed64a15087510604eb5404d8bdafd92e603f5d56660e20ca00b180660"},"schema_version":"1.0","source":{"id":"1908.05546","kind":"arxiv","version":1}},"canonical_sha256":"b694357241700d68a7246311e5c01a1e67708a62b80e27920116d4d5addfce47","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b694357241700d68a7246311e5c01a1e67708a62b80e27920116d4d5addfce47","first_computed_at":"2026-07-04T23:56:53.757810Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-04T23:56:53.757810Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"E0E1ZFSb2Xrk0RonalahmXFnLh0tAo+g22Vn/7T5MLyZ+8pUiC0qEFIUHF2TuehgIkjXoG5t0B5Y3P8FBZ07AA==","signature_status":"signed_v1","signed_at":"2026-07-04T23:56:53.758267Z","signed_message":"canonical_sha256_bytes"},"source_id":"1908.05546","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d212a6cda86188a9545c17cbdf48e4fbbea80de3f9a781edc700ce5e62654d7f","sha256:6e014d3529d41a362c147ef934d5a4695cdde917a370077c6128aaea6e521f39"],"state_sha256":"b94a51af0085dd27b2539ad8eb6fbe926ed260d1fb982df5a0aef8035512170d"}