{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2019:ZUKJXHAMOENJQ4SQF5XRZAYEJW","short_pith_number":"pith:ZUKJXHAM","canonical_record":{"source":{"id":"1912.02877","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-12-05T21:13:36Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"b9d804ec29a78bf6d4406e4c5fd6bc074de535909df229a64ab6f4bead31e31b","abstract_canon_sha256":"53b3db4c031712ea0aa9ede6f8ac7348fdc2ecc7fb474a1597d5803f92fea200"},"schema_version":"1.0"},"canonical_sha256":"cd149b9c0c711a9872502f6f1c83044d8df2fa8645e501ed4720eda533e99b7d","source":{"kind":"arxiv","id":"1912.02877","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1912.02877","created_at":"2026-07-05T03:11:22Z"},{"alias_kind":"arxiv_version","alias_value":"1912.02877v2","created_at":"2026-07-05T03:11:22Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.02877","created_at":"2026-07-05T03:11:22Z"},{"alias_kind":"pith_short_12","alias_value":"ZUKJXHAMOENJ","created_at":"2026-07-05T03:11:22Z"},{"alias_kind":"pith_short_16","alias_value":"ZUKJXHAMOENJQ4SQ","created_at":"2026-07-05T03:11:22Z"},{"alias_kind":"pith_short_8","alias_value":"ZUKJXHAM","created_at":"2026-07-05T03:11:22Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2019:ZUKJXHAMOENJQ4SQF5XRZAYEJW","target":"record","payload":{"canonical_record":{"source":{"id":"1912.02877","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-12-05T21:13:36Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"b9d804ec29a78bf6d4406e4c5fd6bc074de535909df229a64ab6f4bead31e31b","abstract_canon_sha256":"53b3db4c031712ea0aa9ede6f8ac7348fdc2ecc7fb474a1597d5803f92fea200"},"schema_version":"1.0"},"canonical_sha256":"cd149b9c0c711a9872502f6f1c83044d8df2fa8645e501ed4720eda533e99b7d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:11:22.847481Z","signature_b64":"6sogbQdGdfQQXzMoMP9dEYiIQpX1wtwV90r1skkKyxBf4/dFlm2KVhCZD9bRODDwhluOaRc6Box+RFl1GgoKAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd149b9c0c711a9872502f6f1c83044d8df2fa8645e501ed4720eda533e99b7d","last_reissued_at":"2026-07-05T03:11:22.847085Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:11:22.847085Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1912.02877","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:11:22Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uRpTHBSM1fT7rngRTkUGjUtOA3IAHsKDhrw3I+pgYKbtUnhQaIv6HIx7fBPQI0VAEi2qEsJc57jk5wTttXcECA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T14:06:00.869738Z"},"content_sha256":"c1fe2540078542e7131ec77ad3f96159d1656c45675ba64cd5239d11c006ca34","schema_version":"1.0","event_id":"sha256:c1fe2540078542e7131ec77ad3f96159d1656c45675ba64cd5239d11c006ca34"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2019:ZUKJXHAMOENJQ4SQF5XRZAYEJW","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Training Agents using Upside-Down Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Filipe Mutz, J\\\"urgen Schmidhuber, Pranav Shyam, Rupesh Kumar Srivastava, Wojciech Ja\\'skowski","submitted_at":"2019-12-05T21:13:36Z","abstract_excerpt":"We develop Upside-Down Reinforcement Learning (UDRL), a method for learning to act using only supervised learning techniques. Unlike traditional algorithms, UDRL does not use reward prediction or search for an optimal policy. Instead, it trains agents to follow commands such as \"obtain so much total reward in so much time.\" Many of its general principles are outlined in a companion report; the goal of this paper is to develop a practical learning algorithm and show that this conceptually simple perspective on agent training can produce a range of rewarding behaviors for multiple episodic envir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.02877","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.02877/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:11:22Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"binkZg9w+J8wKP+v7hwxlqRV2vMe0ZFS+IgLL1hdT28eN9khszsPTFjSimS6QsU/3mcQEx/CmmOhQcVZglvcCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T14:06:00.870113Z"},"content_sha256":"c1e8b07950687e89126627ad4e1f83f31eeb782b3ec215e50db464c930b82bd3","schema_version":"1.0","event_id":"sha256:c1e8b07950687e89126627ad4e1f83f31eeb782b3ec215e50db464c930b82bd3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/ZUKJXHAMOENJQ4SQF5XRZAYEJW/bundle.json","state_url":"https://pith.science/pith/ZUKJXHAMOENJQ4SQF5XRZAYEJW/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/ZUKJXHAMOENJQ4SQF5XRZAYEJW/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T14:06:00Z","links":{"resolver":"https://pith.science/pith/ZUKJXHAMOENJQ4SQF5XRZAYEJW","bundle":"https://pith.science/pith/ZUKJXHAMOENJQ4SQF5XRZAYEJW/bundle.json","state":"https://pith.science/pith/ZUKJXHAMOENJQ4SQF5XRZAYEJW/state.json","well_known_bundle":"https://pith.science/.well-known/pith/ZUKJXHAMOENJQ4SQF5XRZAYEJW/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:ZUKJXHAMOENJQ4SQF5XRZAYEJW","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"53b3db4c031712ea0aa9ede6f8ac7348fdc2ecc7fb474a1597d5803f92fea200","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-12-05T21:13:36Z","title_canon_sha256":"b9d804ec29a78bf6d4406e4c5fd6bc074de535909df229a64ab6f4bead31e31b"},"schema_version":"1.0","source":{"id":"1912.02877","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1912.02877","created_at":"2026-07-05T03:11:22Z"},{"alias_kind":"arxiv_version","alias_value":"1912.02877v2","created_at":"2026-07-05T03:11:22Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.02877","created_at":"2026-07-05T03:11:22Z"},{"alias_kind":"pith_short_12","alias_value":"ZUKJXHAMOENJ","created_at":"2026-07-05T03:11:22Z"},{"alias_kind":"pith_short_16","alias_value":"ZUKJXHAMOENJQ4SQ","created_at":"2026-07-05T03:11:22Z"},{"alias_kind":"pith_short_8","alias_value":"ZUKJXHAM","created_at":"2026-07-05T03:11:22Z"}],"graph_snapshots":[{"event_id":"sha256:c1e8b07950687e89126627ad4e1f83f31eeb782b3ec215e50db464c930b82bd3","target":"graph","created_at":"2026-07-05T03:11:22Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1912.02877/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We develop Upside-Down Reinforcement Learning (UDRL), a method for learning to act using only supervised learning techniques. Unlike traditional algorithms, UDRL does not use reward prediction or search for an optimal policy. Instead, it trains agents to follow commands such as \"obtain so much total reward in so much time.\" Many of its general principles are outlined in a companion report; the goal of this paper is to develop a practical learning algorithm and show that this conceptually simple perspective on agent training can produce a range of rewarding behaviors for multiple episodic envir","authors_text":"Filipe Mutz, J\\\"urgen Schmidhuber, Pranav Shyam, Rupesh Kumar Srivastava, Wojciech Ja\\'skowski","cross_cats":["cs.AI","cs.RO"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-12-05T21:13:36Z","title":"Training Agents using Upside-Down Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.02877","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c1fe2540078542e7131ec77ad3f96159d1656c45675ba64cd5239d11c006ca34","target":"record","created_at":"2026-07-05T03:11:22Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"53b3db4c031712ea0aa9ede6f8ac7348fdc2ecc7fb474a1597d5803f92fea200","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-12-05T21:13:36Z","title_canon_sha256":"b9d804ec29a78bf6d4406e4c5fd6bc074de535909df229a64ab6f4bead31e31b"},"schema_version":"1.0","source":{"id":"1912.02877","kind":"arxiv","version":2}},"canonical_sha256":"cd149b9c0c711a9872502f6f1c83044d8df2fa8645e501ed4720eda533e99b7d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"cd149b9c0c711a9872502f6f1c83044d8df2fa8645e501ed4720eda533e99b7d","first_computed_at":"2026-07-05T03:11:22.847085Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:11:22.847085Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"6sogbQdGdfQQXzMoMP9dEYiIQpX1wtwV90r1skkKyxBf4/dFlm2KVhCZD9bRODDwhluOaRc6Box+RFl1GgoKAg==","signature_status":"signed_v1","signed_at":"2026-07-05T03:11:22.847481Z","signed_message":"canonical_sha256_bytes"},"source_id":"1912.02877","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c1fe2540078542e7131ec77ad3f96159d1656c45675ba64cd5239d11c006ca34","sha256:c1e8b07950687e89126627ad4e1f83f31eeb782b3ec215e50db464c930b82bd3"],"state_sha256":"2aa70264858de71a97122ef496a706290f0b1455ca2e0dc247cb9cfcaa15e6c8"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uiLbVepUEgiHN1xSD7hixq527J1rL9wcZmig9wZ2zn7ooh7BNBkpq0C+MxDDREMfI3oo1zbK3++w4UP+Gzg5AQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T14:06:00.872397Z","bundle_sha256":"d473850b354fcd7ad1601699074d67f019156719c5db4465e19135561f56f214"}}