{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2020:OSG363736O3SSULZCUQDBHNKVX","short_pith_number":"pith:OSG36373","canonical_record":{"source":{"id":"2011.07553","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2020-11-15T15:25:56Z","cross_cats_sorted":[],"title_canon_sha256":"f2d99bccce49b0dd7338467741ae7239685d175e15f17989e5116ab22e2ed5f9","abstract_canon_sha256":"e950e75842dec7185072094fb370beca1f166daefb85c1912a5ea1ac73e92890"},"schema_version":"1.0"},"canonical_sha256":"748dbf6ffbf3b72951791520309daaadd0bdcaba2c347187fa5393ef6581e240","source":{"kind":"arxiv","id":"2011.07553","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2011.07553","created_at":"2026-07-05T02:27:30Z"},{"alias_kind":"arxiv_version","alias_value":"2011.07553v2","created_at":"2026-07-05T02:27:30Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.07553","created_at":"2026-07-05T02:27:30Z"},{"alias_kind":"pith_short_12","alias_value":"OSG363736O3S","created_at":"2026-07-05T02:27:30Z"},{"alias_kind":"pith_short_16","alias_value":"OSG363736O3SSULZ","created_at":"2026-07-05T02:27:30Z"},{"alias_kind":"pith_short_8","alias_value":"OSG36373","created_at":"2026-07-05T02:27:30Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2020:OSG363736O3SSULZCUQDBHNKVX","target":"record","payload":{"canonical_record":{"source":{"id":"2011.07553","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2020-11-15T15:25:56Z","cross_cats_sorted":[],"title_canon_sha256":"f2d99bccce49b0dd7338467741ae7239685d175e15f17989e5116ab22e2ed5f9","abstract_canon_sha256":"e950e75842dec7185072094fb370beca1f166daefb85c1912a5ea1ac73e92890"},"schema_version":"1.0"},"canonical_sha256":"748dbf6ffbf3b72951791520309daaadd0bdcaba2c347187fa5393ef6581e240","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:27:30.861972Z","signature_b64":"sVplA80x3ZGU3Kjx5KrOZXcPFKTJpZpnCItuolFB+6PxRPAg2q6/Uj1RsgPc2e+qIVWhxZ/nh3YSrUfKFgyjCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"748dbf6ffbf3b72951791520309daaadd0bdcaba2c347187fa5393ef6581e240","last_reissued_at":"2026-07-05T02:27:30.861497Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:27:30.861497Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2011.07553","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:27:30Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"rFOFqXw2UFArGS8BenXchR0PNVHRu+HVpxlR1WIXxP1kAr+Wcff5W4i4+3+1EYctXum0hhafL0pBJkoX9C3xCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T22:02:53.629912Z"},"content_sha256":"330128200ce6c098cb73f0383eba36fdf486ede72928697964e3bbc4e86a271d","schema_version":"1.0","event_id":"sha256:330128200ce6c098cb73f0383eba36fdf486ede72928697964e3bbc4e86a271d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2020:OSG363736O3SSULZCUQDBHNKVX","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"CDT: Cascading Decision Trees for Explainable Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Changjian Li, Gavin Weiguang Ding, Pablo Hernandez-Leal, Ruitong Huang, Zihan Ding","submitted_at":"2020-11-15T15:25:56Z","abstract_excerpt":"Deep Reinforcement Learning (DRL) has recently achieved significant advances in various domains. However, explaining the policy of RL agents still remains an open problem due to several factors, one being the complexity of explaining neural networks decisions. Recently, a group of works have used decision-tree-based models to learn explainable policies. Soft decision trees (SDTs) and discretized differentiable decision trees (DDTs) have been demonstrated to achieve both good performance and share the benefit of having explainable policies. In this work, we further improve the results for tree-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.07553","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.07553/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:27:30Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"aTSt+ZArrA1j9GaTYpyWqPjPMWiZdOd9UjTcDnigf+666FeC6ZjmieDb9Tr1asr59+ZpGku7FNb+Ad6QN02WBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T22:02:53.630787Z"},"content_sha256":"a0a4e0f99de275c4085b2b7eca119a7aa6cb369eddbe05d1cf709a66adb22f54","schema_version":"1.0","event_id":"sha256:a0a4e0f99de275c4085b2b7eca119a7aa6cb369eddbe05d1cf709a66adb22f54"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/OSG363736O3SSULZCUQDBHNKVX/bundle.json","state_url":"https://pith.science/pith/OSG363736O3SSULZCUQDBHNKVX/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/OSG363736O3SSULZCUQDBHNKVX/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T22:02:53Z","links":{"resolver":"https://pith.science/pith/OSG363736O3SSULZCUQDBHNKVX","bundle":"https://pith.science/pith/OSG363736O3SSULZCUQDBHNKVX/bundle.json","state":"https://pith.science/pith/OSG363736O3SSULZCUQDBHNKVX/state.json","well_known_bundle":"https://pith.science/.well-known/pith/OSG363736O3SSULZCUQDBHNKVX/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:OSG363736O3SSULZCUQDBHNKVX","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e950e75842dec7185072094fb370beca1f166daefb85c1912a5ea1ac73e92890","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2020-11-15T15:25:56Z","title_canon_sha256":"f2d99bccce49b0dd7338467741ae7239685d175e15f17989e5116ab22e2ed5f9"},"schema_version":"1.0","source":{"id":"2011.07553","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2011.07553","created_at":"2026-07-05T02:27:30Z"},{"alias_kind":"arxiv_version","alias_value":"2011.07553v2","created_at":"2026-07-05T02:27:30Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.07553","created_at":"2026-07-05T02:27:30Z"},{"alias_kind":"pith_short_12","alias_value":"OSG363736O3S","created_at":"2026-07-05T02:27:30Z"},{"alias_kind":"pith_short_16","alias_value":"OSG363736O3SSULZ","created_at":"2026-07-05T02:27:30Z"},{"alias_kind":"pith_short_8","alias_value":"OSG36373","created_at":"2026-07-05T02:27:30Z"}],"graph_snapshots":[{"event_id":"sha256:a0a4e0f99de275c4085b2b7eca119a7aa6cb369eddbe05d1cf709a66adb22f54","target":"graph","created_at":"2026-07-05T02:27:30Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2011.07553/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Deep Reinforcement Learning (DRL) has recently achieved significant advances in various domains. However, explaining the policy of RL agents still remains an open problem due to several factors, one being the complexity of explaining neural networks decisions. Recently, a group of works have used decision-tree-based models to learn explainable policies. Soft decision trees (SDTs) and discretized differentiable decision trees (DDTs) have been demonstrated to achieve both good performance and share the benefit of having explainable policies. In this work, we further improve the results for tree-","authors_text":"Changjian Li, Gavin Weiguang Ding, Pablo Hernandez-Leal, Ruitong Huang, Zihan Ding","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2020-11-15T15:25:56Z","title":"CDT: Cascading Decision Trees for Explainable Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.07553","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:330128200ce6c098cb73f0383eba36fdf486ede72928697964e3bbc4e86a271d","target":"record","created_at":"2026-07-05T02:27:30Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e950e75842dec7185072094fb370beca1f166daefb85c1912a5ea1ac73e92890","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2020-11-15T15:25:56Z","title_canon_sha256":"f2d99bccce49b0dd7338467741ae7239685d175e15f17989e5116ab22e2ed5f9"},"schema_version":"1.0","source":{"id":"2011.07553","kind":"arxiv","version":2}},"canonical_sha256":"748dbf6ffbf3b72951791520309daaadd0bdcaba2c347187fa5393ef6581e240","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"748dbf6ffbf3b72951791520309daaadd0bdcaba2c347187fa5393ef6581e240","first_computed_at":"2026-07-05T02:27:30.861497Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:27:30.861497Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"sVplA80x3ZGU3Kjx5KrOZXcPFKTJpZpnCItuolFB+6PxRPAg2q6/Uj1RsgPc2e+qIVWhxZ/nh3YSrUfKFgyjCA==","signature_status":"signed_v1","signed_at":"2026-07-05T02:27:30.861972Z","signed_message":"canonical_sha256_bytes"},"source_id":"2011.07553","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:330128200ce6c098cb73f0383eba36fdf486ede72928697964e3bbc4e86a271d","sha256:a0a4e0f99de275c4085b2b7eca119a7aa6cb369eddbe05d1cf709a66adb22f54"],"state_sha256":"5301c6baabe70069cdfce889105d444c4b721f36e7cf4af6aff241f347a4bb2a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"h7xLZR3GqkxcXO4DzSV+HjRdSRWOW/1b6SyuzJtbezxC5/VczTlZYac9z/rD/wBSVDvsyHyITxvTr9nqPMntCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T22:02:53.637383Z","bundle_sha256":"4253cf329577e49ebbe8d7b65a7dc218d4e6ab4e30f1c96a1841933cbc513553"}}