{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:HRLVWW3WNRSOFQMK5SBLBZG5KS","short_pith_number":"pith:HRLVWW3W","schema_version":"1.0","canonical_sha256":"3c575b5b766c64e2c18aec82b0e4dd54a71147d95fb687a36b9391fb1c4f3434","source":{"kind":"arxiv","id":"2002.03072","version":1},"attestation_state":"computed","paper":{"title":"Generalized Hidden Parameter MDPs Transferable Model-based RL in a Handful of Trials","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Christian F. Perez, Felipe Petroski Such, Theofanis Karaletsos","submitted_at":"2020-02-08T02:49:33Z","abstract_excerpt":"There is broad interest in creating RL agents that can solve many (related) tasks and adapt to new tasks and environments after initial training. Model-based RL leverages learned surrogate models that describe dynamics and rewards of individual tasks, such that planning in a good surrogate can lead to good control of the true system. Rather than solving each task individually from scratch, hierarchical models can exploit the fact that tasks are often related by (unobserved) causal factors of variation in order to achieve efficient generalization, as in learning how the mass of an item affects "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.03072","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-08T02:49:33Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"7170be41fea8d383ce750c1a67cfde6a8e6cb307df6a05b14cdc4d66688d3173","abstract_canon_sha256":"54da42f9f41a671b26060079e71d94437020180dba1acb9539b81f958d6b6f22"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:39:18.621474Z","signature_b64":"8DZwi0wP7ejkGgtzXLKH4r8YQZZDglLA194bbl6hR8F17Z4S7OxuCeJRHdS8bYkXF0edsH1gyT1GzK+kRqbPAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c575b5b766c64e2c18aec82b0e4dd54a71147d95fb687a36b9391fb1c4f3434","last_reissued_at":"2026-07-05T00:39:18.621048Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:39:18.621048Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalized Hidden Parameter MDPs Transferable Model-based RL in a Handful of Trials","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Christian F. Perez, Felipe Petroski Such, Theofanis Karaletsos","submitted_at":"2020-02-08T02:49:33Z","abstract_excerpt":"There is broad interest in creating RL agents that can solve many (related) tasks and adapt to new tasks and environments after initial training. Model-based RL leverages learned surrogate models that describe dynamics and rewards of individual tasks, such that planning in a good surrogate can lead to good control of the true system. Rather than solving each task individually from scratch, hierarchical models can exploit the fact that tasks are often related by (unobserved) causal factors of variation in order to achieve efficient generalization, as in learning how the mass of an item affects "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.03072","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.03072/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.03072","created_at":"2026-07-05T00:39:18.621108+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.03072v1","created_at":"2026-07-05T00:39:18.621108+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.03072","created_at":"2026-07-05T00:39:18.621108+00:00"},{"alias_kind":"pith_short_12","alias_value":"HRLVWW3WNRSO","created_at":"2026-07-05T00:39:18.621108+00:00"},{"alias_kind":"pith_short_16","alias_value":"HRLVWW3WNRSOFQMK","created_at":"2026-07-05T00:39:18.621108+00:00"},{"alias_kind":"pith_short_8","alias_value":"HRLVWW3W","created_at":"2026-07-05T00:39:18.621108+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HRLVWW3WNRSOFQMK5SBLBZG5KS","json":"https://pith.science/pith/HRLVWW3WNRSOFQMK5SBLBZG5KS.json","graph_json":"https://pith.science/api/pith-number/HRLVWW3WNRSOFQMK5SBLBZG5KS/graph.json","events_json":"https://pith.science/api/pith-number/HRLVWW3WNRSOFQMK5SBLBZG5KS/events.json","paper":"https://pith.science/paper/HRLVWW3W"},"agent_actions":{"view_html":"https://pith.science/pith/HRLVWW3WNRSOFQMK5SBLBZG5KS","download_json":"https://pith.science/pith/HRLVWW3WNRSOFQMK5SBLBZG5KS.json","view_paper":"https://pith.science/paper/HRLVWW3W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.03072&json=true","fetch_graph":"https://pith.science/api/pith-number/HRLVWW3WNRSOFQMK5SBLBZG5KS/graph.json","fetch_events":"https://pith.science/api/pith-number/HRLVWW3WNRSOFQMK5SBLBZG5KS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HRLVWW3WNRSOFQMK5SBLBZG5KS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HRLVWW3WNRSOFQMK5SBLBZG5KS/action/storage_attestation","attest_author":"https://pith.science/pith/HRLVWW3WNRSOFQMK5SBLBZG5KS/action/author_attestation","sign_citation":"https://pith.science/pith/HRLVWW3WNRSOFQMK5SBLBZG5KS/action/citation_signature","submit_replication":"https://pith.science/pith/HRLVWW3WNRSOFQMK5SBLBZG5KS/action/replication_record"}},"created_at":"2026-07-05T00:39:18.621108+00:00","updated_at":"2026-07-05T00:39:18.621108+00:00"}