{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2018:DXBSAHOUJFSL5N7ROJCQXFZFT2","short_pith_number":"pith:DXBSAHOU","canonical_record":{"source":{"id":"1805.10129","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-23T23:23:34Z","cross_cats_sorted":["cs.AI","cs.CV","stat.ML"],"title_canon_sha256":"e9782200e524d6459cb1c678d612ee2691f5b42ce1278d1459ae9cd04650dea9","abstract_canon_sha256":"d7cd93b07c64e6af495f350bde1c50375c3cb535523e4c8f4a3813750b7c1156"},"schema_version":"1.0"},"canonical_sha256":"1dc3201dd44964beb7f172450b97259e8c80e608f5a45554abb3bcd38bf0c9c7","source":{"kind":"arxiv","id":"1805.10129","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1805.10129","created_at":"2026-05-18T00:14:58Z"},{"alias_kind":"arxiv_version","alias_value":"1805.10129v1","created_at":"2026-05-18T00:14:58Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1805.10129","created_at":"2026-05-18T00:14:58Z"},{"alias_kind":"pith_short_12","alias_value":"DXBSAHOUJFSL","created_at":"2026-05-18T12:32:19Z"},{"alias_kind":"pith_short_16","alias_value":"DXBSAHOUJFSL5N7R","created_at":"2026-05-18T12:32:19Z"},{"alias_kind":"pith_short_8","alias_value":"DXBSAHOU","created_at":"2026-05-18T12:32:19Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2018:DXBSAHOUJFSL5N7ROJCQXFZFT2","target":"record","payload":{"canonical_record":{"source":{"id":"1805.10129","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-23T23:23:34Z","cross_cats_sorted":["cs.AI","cs.CV","stat.ML"],"title_canon_sha256":"e9782200e524d6459cb1c678d612ee2691f5b42ce1278d1459ae9cd04650dea9","abstract_canon_sha256":"d7cd93b07c64e6af495f350bde1c50375c3cb535523e4c8f4a3813750b7c1156"},"schema_version":"1.0"},"canonical_sha256":"1dc3201dd44964beb7f172450b97259e8c80e608f5a45554abb3bcd38bf0c9c7","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:14:58.200781Z","signature_b64":"nU30XwYboeQ8+hUnSsd1GbWFGSPBbwq7y38W32txhw0eFqS46xqzE99O9XU9qobeo5OxhrcJcs7TPhps1QzkAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1dc3201dd44964beb7f172450b97259e8c80e608f5a45554abb3bcd38bf0c9c7","last_reissued_at":"2026-05-18T00:14:58.200074Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:14:58.200074Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1805.10129","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:14:58Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"eUL+y9aDhUMi+j0VSz1Ep0KGwxPGOIyeXEMRsx/PRt0oEzp07h6bH0nwRcEL48yKTxkiJtheGpDla6FxrnBICA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-25T17:33:46.335107Z"},"content_sha256":"ef9fa71c7c566f86132bdda271a1483b4a7d90d80241bf040cea8996e06f853f","schema_version":"1.0","event_id":"sha256:ef9fa71c7c566f86132bdda271a1483b4a7d90d80241bf040cea8996e06f853f"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2018:DXBSAHOUJFSL5N7ROJCQXFZFT2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Dyna Planning using a Feature Based Generative Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Doina Precup, Ryan Faulkner","submitted_at":"2018-05-23T23:23:34Z","abstract_excerpt":"Dyna-style reinforcement learning is a powerful approach for problems where not much real data is available. The main idea is to supplement real trajectories, or sequences of sampled states over time, with simulated ones sampled from a learned model of the environment. However, in large state spaces, the problem of learning a good generative model of the environment has been open so far. We propose to use deep belief networks to learn an environment model for use in Dyna. We present our approach and validate it empirically on problems where the state observations consist of images. Our results"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1805.10129","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:14:58Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PdqDwFQO03A6ZIUJmzXCDqg3TrixXrj+voC8+tYIL/kgW0waOt28LMX0GBNNoFJFaHPFSMLnnhDsdCHDXa4GDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-25T17:33:46.335851Z"},"content_sha256":"909479af2320a54860fdf3111a2facad5233af2698d79be2a1e5a4568ec17d59","schema_version":"1.0","event_id":"sha256:909479af2320a54860fdf3111a2facad5233af2698d79be2a1e5a4568ec17d59"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/DXBSAHOUJFSL5N7ROJCQXFZFT2/bundle.json","state_url":"https://pith.science/pith/DXBSAHOUJFSL5N7ROJCQXFZFT2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/DXBSAHOUJFSL5N7ROJCQXFZFT2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-25T17:33:46Z","links":{"resolver":"https://pith.science/pith/DXBSAHOUJFSL5N7ROJCQXFZFT2","bundle":"https://pith.science/pith/DXBSAHOUJFSL5N7ROJCQXFZFT2/bundle.json","state":"https://pith.science/pith/DXBSAHOUJFSL5N7ROJCQXFZFT2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/DXBSAHOUJFSL5N7ROJCQXFZFT2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2018:DXBSAHOUJFSL5N7ROJCQXFZFT2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d7cd93b07c64e6af495f350bde1c50375c3cb535523e4c8f4a3813750b7c1156","cross_cats_sorted":["cs.AI","cs.CV","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-23T23:23:34Z","title_canon_sha256":"e9782200e524d6459cb1c678d612ee2691f5b42ce1278d1459ae9cd04650dea9"},"schema_version":"1.0","source":{"id":"1805.10129","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1805.10129","created_at":"2026-05-18T00:14:58Z"},{"alias_kind":"arxiv_version","alias_value":"1805.10129v1","created_at":"2026-05-18T00:14:58Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1805.10129","created_at":"2026-05-18T00:14:58Z"},{"alias_kind":"pith_short_12","alias_value":"DXBSAHOUJFSL","created_at":"2026-05-18T12:32:19Z"},{"alias_kind":"pith_short_16","alias_value":"DXBSAHOUJFSL5N7R","created_at":"2026-05-18T12:32:19Z"},{"alias_kind":"pith_short_8","alias_value":"DXBSAHOU","created_at":"2026-05-18T12:32:19Z"}],"graph_snapshots":[{"event_id":"sha256:909479af2320a54860fdf3111a2facad5233af2698d79be2a1e5a4568ec17d59","target":"graph","created_at":"2026-05-18T00:14:58Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"Dyna-style reinforcement learning is a powerful approach for problems where not much real data is available. The main idea is to supplement real trajectories, or sequences of sampled states over time, with simulated ones sampled from a learned model of the environment. However, in large state spaces, the problem of learning a good generative model of the environment has been open so far. We propose to use deep belief networks to learn an environment model for use in Dyna. We present our approach and validate it empirically on problems where the state observations consist of images. Our results","authors_text":"Doina Precup, Ryan Faulkner","cross_cats":["cs.AI","cs.CV","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-23T23:23:34Z","title":"Dyna Planning using a Feature Based Generative Model"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1805.10129","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ef9fa71c7c566f86132bdda271a1483b4a7d90d80241bf040cea8996e06f853f","target":"record","created_at":"2026-05-18T00:14:58Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d7cd93b07c64e6af495f350bde1c50375c3cb535523e4c8f4a3813750b7c1156","cross_cats_sorted":["cs.AI","cs.CV","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-23T23:23:34Z","title_canon_sha256":"e9782200e524d6459cb1c678d612ee2691f5b42ce1278d1459ae9cd04650dea9"},"schema_version":"1.0","source":{"id":"1805.10129","kind":"arxiv","version":1}},"canonical_sha256":"1dc3201dd44964beb7f172450b97259e8c80e608f5a45554abb3bcd38bf0c9c7","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1dc3201dd44964beb7f172450b97259e8c80e608f5a45554abb3bcd38bf0c9c7","first_computed_at":"2026-05-18T00:14:58.200074Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T00:14:58.200074Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"nU30XwYboeQ8+hUnSsd1GbWFGSPBbwq7y38W32txhw0eFqS46xqzE99O9XU9qobeo5OxhrcJcs7TPhps1QzkAA==","signature_status":"signed_v1","signed_at":"2026-05-18T00:14:58.200781Z","signed_message":"canonical_sha256_bytes"},"source_id":"1805.10129","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ef9fa71c7c566f86132bdda271a1483b4a7d90d80241bf040cea8996e06f853f","sha256:909479af2320a54860fdf3111a2facad5233af2698d79be2a1e5a4568ec17d59"],"state_sha256":"98f0463cdf10b799cf56001262018936374819ee3e98592d0f2cc058587a8472"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"R/jZK3VT60TM7tKOCKXfP497yRF+sjtfCUgVIpICTYXlqXHO4WpnS+ijZwKMPF3MSrZdU+4Dn1QHb55BiIZXBw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-25T17:33:46.339781Z","bundle_sha256":"3ec109471d3f99cf9286251acd5dbc7433789d19e95178c77a15234c97d2093f"}}