{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:RFME6KYUGJXUIHAPCA6OKULXI7","short_pith_number":"pith:RFME6KYU","schema_version":"1.0","canonical_sha256":"89584f2b14326f441c0f103ce5517747f7ac002eeb7364e7182a3ff8a3f9fe38","source":{"kind":"arxiv","id":"1912.02875","version":2},"attestation_state":"computed","paper":{"title":"Reinforcement Learning Upside Down: Don't Predict Rewards -- Just Map Them to Actions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Juergen Schmidhuber","submitted_at":"2019-12-05T21:10:08Z","abstract_excerpt":"We transform reinforcement learning (RL) into a form of supervised learning (SL) by turning traditional RL on its head, calling this Upside Down RL (UDRL). Standard RL predicts rewards, while UDRL instead uses rewards as task-defining inputs, together with representations of time horizons and other computable functions of historic and desired future data. UDRL learns to interpret these input observations as commands, mapping them to actions (or action probabilities) through SL on past (possibly accidental) experience. UDRL generalizes to achieve high rewards or other goals, through input comma"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.02875","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2019-12-05T21:10:08Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ef0265bf1724762dec779753e2c9ef82b51dcc8975f1a3ddd69ac5b6d8fc6f8a","abstract_canon_sha256":"82b3905583ec42d98e915d113a3ec763d970e104fdaca11ad2f8f6dfb655fd6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:12:29.756856Z","signature_b64":"voyS/9V6cOnZiogHPj8h3xhZ5sPTfndlHapgb1mISW5jep0tTlvhoNoNhrwlCrT8YRTc/ZtTh43pHarvOvTuBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89584f2b14326f441c0f103ce5517747f7ac002eeb7364e7182a3ff8a3f9fe38","last_reissued_at":"2026-07-05T01:12:29.756400Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:12:29.756400Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning Upside Down: Don't Predict Rewards -- Just Map Them to Actions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Juergen Schmidhuber","submitted_at":"2019-12-05T21:10:08Z","abstract_excerpt":"We transform reinforcement learning (RL) into a form of supervised learning (SL) by turning traditional RL on its head, calling this Upside Down RL (UDRL). Standard RL predicts rewards, while UDRL instead uses rewards as task-defining inputs, together with representations of time horizons and other computable functions of historic and desired future data. UDRL learns to interpret these input observations as commands, mapping them to actions (or action probabilities) through SL on past (possibly accidental) experience. UDRL generalizes to achieve high rewards or other goals, through input comma"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.02875","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.02875/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.02875","created_at":"2026-07-05T01:12:29.756455+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.02875v2","created_at":"2026-07-05T01:12:29.756455+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.02875","created_at":"2026-07-05T01:12:29.756455+00:00"},{"alias_kind":"pith_short_12","alias_value":"RFME6KYUGJXU","created_at":"2026-07-05T01:12:29.756455+00:00"},{"alias_kind":"pith_short_16","alias_value":"RFME6KYUGJXUIHAP","created_at":"2026-07-05T01:12:29.756455+00:00"},{"alias_kind":"pith_short_8","alias_value":"RFME6KYU","created_at":"2026-07-05T01:12:29.756455+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24231","citing_title":"FlowR2A: Learning Reward-to-Action Distribution for Multimodal Driving Planning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32027","citing_title":"Freeform Preference Learning for Robotic Manipulation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2506.10137","citing_title":"Self-Predictive Representations for Combinatorial Generalization in Behavioral Cloning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2211.15657","citing_title":"Is Conditional Generative Modeling all you need for Decision-Making?","ref_index":243,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14759","citing_title":"$\\pi^{*}_{0.6}$: a VLA That Learns From Experience","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":168,"is_internal_anchor":false},{"citing_arxiv_id":"2308.00352","citing_title":"MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework","ref_index":151,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RFME6KYUGJXUIHAPCA6OKULXI7","json":"https://pith.science/pith/RFME6KYUGJXUIHAPCA6OKULXI7.json","graph_json":"https://pith.science/api/pith-number/RFME6KYUGJXUIHAPCA6OKULXI7/graph.json","events_json":"https://pith.science/api/pith-number/RFME6KYUGJXUIHAPCA6OKULXI7/events.json","paper":"https://pith.science/paper/RFME6KYU"},"agent_actions":{"view_html":"https://pith.science/pith/RFME6KYUGJXUIHAPCA6OKULXI7","download_json":"https://pith.science/pith/RFME6KYUGJXUIHAPCA6OKULXI7.json","view_paper":"https://pith.science/paper/RFME6KYU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.02875&json=true","fetch_graph":"https://pith.science/api/pith-number/RFME6KYUGJXUIHAPCA6OKULXI7/graph.json","fetch_events":"https://pith.science/api/pith-number/RFME6KYUGJXUIHAPCA6OKULXI7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RFME6KYUGJXUIHAPCA6OKULXI7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RFME6KYUGJXUIHAPCA6OKULXI7/action/storage_attestation","attest_author":"https://pith.science/pith/RFME6KYUGJXUIHAPCA6OKULXI7/action/author_attestation","sign_citation":"https://pith.science/pith/RFME6KYUGJXUIHAPCA6OKULXI7/action/citation_signature","submit_replication":"https://pith.science/pith/RFME6KYUGJXUIHAPCA6OKULXI7/action/replication_record"}},"created_at":"2026-07-05T01:12:29.756455+00:00","updated_at":"2026-07-05T01:12:29.756455+00:00"}