{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:GD5TQCRAI7G442DRWJUPJGL56P","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d5d2339a65a547b9b4ca593899f1735df361816668169e200f320b8b5fbc1790","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-03T20:38:27Z","title_canon_sha256":"2a6e1573553592e881931a087245b20c4d77fb8402af6ec93103e37da266889b"},"schema_version":"1.0","source":{"id":"2206.01812","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.01812","created_at":"2026-07-05T04:28:56Z"},{"alias_kind":"arxiv_version","alias_value":"2206.01812v1","created_at":"2026-07-05T04:28:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.01812","created_at":"2026-07-05T04:28:56Z"},{"alias_kind":"pith_short_12","alias_value":"GD5TQCRAI7G4","created_at":"2026-07-05T04:28:56Z"},{"alias_kind":"pith_short_16","alias_value":"GD5TQCRAI7G442DR","created_at":"2026-07-05T04:28:56Z"},{"alias_kind":"pith_short_8","alias_value":"GD5TQCRA","created_at":"2026-07-05T04:28:56Z"}],"graph_snapshots":[{"event_id":"sha256:b185f99fac3b34e11068ea553eb7e46cc16621373476d2661100fcb3f0a9fb5c","target":"graph","created_at":"2026-07-05T04:28:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2206.01812/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Deep reinforcement learning has shown promise in discrete domains requiring complex reasoning, including games such as Chess, Go, and Hanabi. However, this type of reasoning is less often observed in long-horizon, continuous domains with high-dimensional observations, where instead RL research has predominantly focused on problems with simple high-level structure (e.g. opening a drawer or moving a robot as fast as possible). Inspired by combinatorially hard optimization problems, we propose a set of robotics tasks which admit many distinct solutions at the high-level, but require reasoning abo","authors_text":"Andrew C. Li, Pashootan Vaezipoor, Rodrigo Toro Icarte, Sheila A. McIlraith","cross_cats":["cs.AI","cs.RO"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-03T20:38:27Z","title":"Challenges to Solving Combinatorially Hard Long-Horizon Deep RL Tasks"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.01812","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d4e107fb071f8c560e7470d2e0f92a10243473126b44a6e53adc7e23e9d602ca","target":"record","created_at":"2026-07-05T04:28:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d5d2339a65a547b9b4ca593899f1735df361816668169e200f320b8b5fbc1790","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-03T20:38:27Z","title_canon_sha256":"2a6e1573553592e881931a087245b20c4d77fb8402af6ec93103e37da266889b"},"schema_version":"1.0","source":{"id":"2206.01812","kind":"arxiv","version":1}},"canonical_sha256":"30fb380a2047cdce6871b268f4997df3ffeaab5cf87c15acb897bac59f6966d5","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"30fb380a2047cdce6871b268f4997df3ffeaab5cf87c15acb897bac59f6966d5","first_computed_at":"2026-07-05T04:28:56.451391Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T04:28:56.451391Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"4ppVQCNwBj8RB5RKS2Xp2dCe/WWt4oAWzXOxp5NTY8BLngY8MJpVz/agZDFmzDEt/ND0JJpMcQ9+gJwiZLuZCg==","signature_status":"signed_v1","signed_at":"2026-07-05T04:28:56.451939Z","signed_message":"canonical_sha256_bytes"},"source_id":"2206.01812","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d4e107fb071f8c560e7470d2e0f92a10243473126b44a6e53adc7e23e9d602ca","sha256:b185f99fac3b34e11068ea553eb7e46cc16621373476d2661100fcb3f0a9fb5c"],"state_sha256":"90cb2b61956c1d93fcdafdf4f97050c992464f0a1bf015e5f8f1a339a16409b6"}