{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:MZK2Q2Q7DRLQL4YVASVBSRIV6B","short_pith_number":"pith:MZK2Q2Q7","canonical_record":{"source":{"id":"2210.13435","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-24T17:49:56Z","cross_cats_sorted":[],"title_canon_sha256":"b5e668b93e501cff2c7f2bc58dbb811a9a976ca09628b9e1f8be6399e1e9390d","abstract_canon_sha256":"02228b0e7f7464d84c6134f2e7d163fef000e448bbe330f59a96385c5cb7dc32"},"schema_version":"1.0"},"canonical_sha256":"6655a86a1f1c5705f31504aa194515f06462d8d545747e6d9f1b29eea730f699","source":{"kind":"arxiv","id":"2210.13435","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.13435","created_at":"2026-07-05T05:09:45Z"},{"alias_kind":"arxiv_version","alias_value":"2210.13435v1","created_at":"2026-07-05T05:09:45Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.13435","created_at":"2026-07-05T05:09:45Z"},{"alias_kind":"pith_short_12","alias_value":"MZK2Q2Q7DRLQ","created_at":"2026-07-05T05:09:45Z"},{"alias_kind":"pith_short_16","alias_value":"MZK2Q2Q7DRLQL4YV","created_at":"2026-07-05T05:09:45Z"},{"alias_kind":"pith_short_8","alias_value":"MZK2Q2Q7","created_at":"2026-07-05T05:09:45Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:MZK2Q2Q7DRLQL4YVASVBSRIV6B","target":"record","payload":{"canonical_record":{"source":{"id":"2210.13435","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-24T17:49:56Z","cross_cats_sorted":[],"title_canon_sha256":"b5e668b93e501cff2c7f2bc58dbb811a9a976ca09628b9e1f8be6399e1e9390d","abstract_canon_sha256":"02228b0e7f7464d84c6134f2e7d163fef000e448bbe330f59a96385c5cb7dc32"},"schema_version":"1.0"},"canonical_sha256":"6655a86a1f1c5705f31504aa194515f06462d8d545747e6d9f1b29eea730f699","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:09:45.400337Z","signature_b64":"z068rnGQDM+/t6M8DgcMe3uhdsyUq+gug1KrMtsnx6Zk1ZSnUx4vfh4kqDl/vngq0KKI7ZVDBu2rTQL+ExInBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6655a86a1f1c5705f31504aa194515f06462d8d545747e6d9f1b29eea730f699","last_reissued_at":"2026-07-05T05:09:45.399819Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:09:45.399819Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2210.13435","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:09:45Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2Zkx9Mio+26zfdPplejRtXFr3rJRg2uHnG/RLQUs71inMUcj2hZFhNVJbUv6eHUnhvejcJOh7dLuNafg5GrBAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T21:35:30.053313Z"},"content_sha256":"593d534bb80bba26529014fa9bb829f3c773794a3c89e537710895fb838293e6","schema_version":"1.0","event_id":"sha256:593d534bb80bba26529014fa9bb829f3c773794a3c89e537710895fb838293e6"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:MZK2Q2Q7DRLQL4YVASVBSRIV6B","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Dichotomy of Control: Separating What You Can Control from What You Cannot","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dale Schuurmans, Mengjiao Yang, Ofir Nachum, Pieter Abbeel","submitted_at":"2022-10-24T17:49:56Z","abstract_excerpt":"Future- or return-conditioned supervised learning is an emerging paradigm for offline reinforcement learning (RL), where the future outcome (i.e., return) associated with an observed action sequence is used as input to a policy trained to imitate those same actions. While return-conditioning is at the heart of popular algorithms such as decision transformer (DT), these methods tend to perform poorly in highly stochastic environments, where an occasional high return can arise from randomness in the environment rather than the actions themselves. Such situations can lead to a learned policy that"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.13435","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.13435/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:09:45Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CcCT8GtsmjLPjIAsRjDBeyGZjVEY0egbaTY8D6So2KPmMI4sW+ik2o8nYKzCg3+rmUI+JTKG6Bcwa+qrM/6mCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T21:35:30.054784Z"},"content_sha256":"12a9fcad38a5f953d79373783c109699355e9c23b5de7a9fd1ec1c53caf3c412","schema_version":"1.0","event_id":"sha256:12a9fcad38a5f953d79373783c109699355e9c23b5de7a9fd1ec1c53caf3c412"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/MZK2Q2Q7DRLQL4YVASVBSRIV6B/bundle.json","state_url":"https://pith.science/pith/MZK2Q2Q7DRLQL4YVASVBSRIV6B/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/MZK2Q2Q7DRLQL4YVASVBSRIV6B/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T21:35:30Z","links":{"resolver":"https://pith.science/pith/MZK2Q2Q7DRLQL4YVASVBSRIV6B","bundle":"https://pith.science/pith/MZK2Q2Q7DRLQL4YVASVBSRIV6B/bundle.json","state":"https://pith.science/pith/MZK2Q2Q7DRLQL4YVASVBSRIV6B/state.json","well_known_bundle":"https://pith.science/.well-known/pith/MZK2Q2Q7DRLQL4YVASVBSRIV6B/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:MZK2Q2Q7DRLQL4YVASVBSRIV6B","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"02228b0e7f7464d84c6134f2e7d163fef000e448bbe330f59a96385c5cb7dc32","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-24T17:49:56Z","title_canon_sha256":"b5e668b93e501cff2c7f2bc58dbb811a9a976ca09628b9e1f8be6399e1e9390d"},"schema_version":"1.0","source":{"id":"2210.13435","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.13435","created_at":"2026-07-05T05:09:45Z"},{"alias_kind":"arxiv_version","alias_value":"2210.13435v1","created_at":"2026-07-05T05:09:45Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.13435","created_at":"2026-07-05T05:09:45Z"},{"alias_kind":"pith_short_12","alias_value":"MZK2Q2Q7DRLQ","created_at":"2026-07-05T05:09:45Z"},{"alias_kind":"pith_short_16","alias_value":"MZK2Q2Q7DRLQL4YV","created_at":"2026-07-05T05:09:45Z"},{"alias_kind":"pith_short_8","alias_value":"MZK2Q2Q7","created_at":"2026-07-05T05:09:45Z"}],"graph_snapshots":[{"event_id":"sha256:12a9fcad38a5f953d79373783c109699355e9c23b5de7a9fd1ec1c53caf3c412","target":"graph","created_at":"2026-07-05T05:09:45Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2210.13435/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Future- or return-conditioned supervised learning is an emerging paradigm for offline reinforcement learning (RL), where the future outcome (i.e., return) associated with an observed action sequence is used as input to a policy trained to imitate those same actions. While return-conditioning is at the heart of popular algorithms such as decision transformer (DT), these methods tend to perform poorly in highly stochastic environments, where an occasional high return can arise from randomness in the environment rather than the actions themselves. Such situations can lead to a learned policy that","authors_text":"Dale Schuurmans, Mengjiao Yang, Ofir Nachum, Pieter Abbeel","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-24T17:49:56Z","title":"Dichotomy of Control: Separating What You Can Control from What You Cannot"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.13435","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:593d534bb80bba26529014fa9bb829f3c773794a3c89e537710895fb838293e6","target":"record","created_at":"2026-07-05T05:09:45Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"02228b0e7f7464d84c6134f2e7d163fef000e448bbe330f59a96385c5cb7dc32","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-24T17:49:56Z","title_canon_sha256":"b5e668b93e501cff2c7f2bc58dbb811a9a976ca09628b9e1f8be6399e1e9390d"},"schema_version":"1.0","source":{"id":"2210.13435","kind":"arxiv","version":1}},"canonical_sha256":"6655a86a1f1c5705f31504aa194515f06462d8d545747e6d9f1b29eea730f699","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6655a86a1f1c5705f31504aa194515f06462d8d545747e6d9f1b29eea730f699","first_computed_at":"2026-07-05T05:09:45.399819Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:09:45.399819Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"z068rnGQDM+/t6M8DgcMe3uhdsyUq+gug1KrMtsnx6Zk1ZSnUx4vfh4kqDl/vngq0KKI7ZVDBu2rTQL+ExInBw==","signature_status":"signed_v1","signed_at":"2026-07-05T05:09:45.400337Z","signed_message":"canonical_sha256_bytes"},"source_id":"2210.13435","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:593d534bb80bba26529014fa9bb829f3c773794a3c89e537710895fb838293e6","sha256:12a9fcad38a5f953d79373783c109699355e9c23b5de7a9fd1ec1c53caf3c412"],"state_sha256":"14bff23a91d511e0af24dc2ce0d3d46c659b644e65b75c0703c169bbfe3f7790"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/xchB6kk5mMRfz84dYY2XuorOsRCb6rLJmiRPGjdBAigXzSOOc0HGklSEwtCPv+bFRSRazvq+MZHOAlwNiIPAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T21:35:30.061911Z","bundle_sha256":"13a673ef999051000eae25aea16ba638bc068d43703aee93ebd40d0021a5cd9c"}}