{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:RFME6KYUGJXUIHAPCA6OKULXI7","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"82b3905583ec42d98e915d113a3ec763d970e104fdaca11ad2f8f6dfb655fd6a","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2019-12-05T21:10:08Z","title_canon_sha256":"ef0265bf1724762dec779753e2c9ef82b51dcc8975f1a3ddd69ac5b6d8fc6f8a"},"schema_version":"1.0","source":{"id":"1912.02875","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1912.02875","created_at":"2026-07-05T01:12:29Z"},{"alias_kind":"arxiv_version","alias_value":"1912.02875v2","created_at":"2026-07-05T01:12:29Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.02875","created_at":"2026-07-05T01:12:29Z"},{"alias_kind":"pith_short_12","alias_value":"RFME6KYUGJXU","created_at":"2026-07-05T01:12:29Z"},{"alias_kind":"pith_short_16","alias_value":"RFME6KYUGJXUIHAP","created_at":"2026-07-05T01:12:29Z"},{"alias_kind":"pith_short_8","alias_value":"RFME6KYU","created_at":"2026-07-05T01:12:29Z"}],"graph_snapshots":[{"event_id":"sha256:6b198d49c31beee1841c9235bad41f2c5c052deac52274970e4fc63c7e23a19f","target":"graph","created_at":"2026-07-05T01:12:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1912.02875/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We transform reinforcement learning (RL) into a form of supervised learning (SL) by turning traditional RL on its head, calling this Upside Down RL (UDRL). Standard RL predicts rewards, while UDRL instead uses rewards as task-defining inputs, together with representations of time horizons and other computable functions of historic and desired future data. UDRL learns to interpret these input observations as commands, mapping them to actions (or action probabilities) through SL on past (possibly accidental) experience. UDRL generalizes to achieve high rewards or other goals, through input comma","authors_text":"Juergen Schmidhuber","cross_cats":["cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2019-12-05T21:10:08Z","title":"Reinforcement Learning Upside Down: Don't Predict Rewards -- Just Map Them to Actions"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.02875","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ad41ed8f455f3c8d829e633fa5bc3bb44f9a0093f66f63b808c29f0f65981bdb","target":"record","created_at":"2026-07-05T01:12:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"82b3905583ec42d98e915d113a3ec763d970e104fdaca11ad2f8f6dfb655fd6a","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2019-12-05T21:10:08Z","title_canon_sha256":"ef0265bf1724762dec779753e2c9ef82b51dcc8975f1a3ddd69ac5b6d8fc6f8a"},"schema_version":"1.0","source":{"id":"1912.02875","kind":"arxiv","version":2}},"canonical_sha256":"89584f2b14326f441c0f103ce5517747f7ac002eeb7364e7182a3ff8a3f9fe38","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"89584f2b14326f441c0f103ce5517747f7ac002eeb7364e7182a3ff8a3f9fe38","first_computed_at":"2026-07-05T01:12:29.756400Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T01:12:29.756400Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"voyS/9V6cOnZiogHPj8h3xhZ5sPTfndlHapgb1mISW5jep0tTlvhoNoNhrwlCrT8YRTc/ZtTh43pHarvOvTuBw==","signature_status":"signed_v1","signed_at":"2026-07-05T01:12:29.756856Z","signed_message":"canonical_sha256_bytes"},"source_id":"1912.02875","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ad41ed8f455f3c8d829e633fa5bc3bb44f9a0093f66f63b808c29f0f65981bdb","sha256:6b198d49c31beee1841c9235bad41f2c5c052deac52274970e4fc63c7e23a19f"],"state_sha256":"90fd322d6eafeea12aa1776f098614dbfd1c4b86eb6e714ed7c1fafeaf6fb7be"}