{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:6NSCU2L3X2KNIA7DI4RWUAT53O","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"a538e72c701d9aab204bcdffb6da29dd8355a9dd777b140f6110a429ac0f4294","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-03-22T03:19:26Z","title_canon_sha256":"d863f4c87a8c52d4a13e6d585b6bf6bcb607db21f42e19dcd3b12d209534339d"},"schema_version":"1.0","source":{"id":"1903.09338","kind":"arxiv","version":5}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1903.09338","created_at":"2026-07-05T01:13:34Z"},{"alias_kind":"arxiv_version","alias_value":"1903.09338v5","created_at":"2026-07-05T01:13:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1903.09338","created_at":"2026-07-05T01:13:34Z"},{"alias_kind":"pith_short_12","alias_value":"6NSCU2L3X2KN","created_at":"2026-07-05T01:13:34Z"},{"alias_kind":"pith_short_16","alias_value":"6NSCU2L3X2KNIA7D","created_at":"2026-07-05T01:13:34Z"},{"alias_kind":"pith_short_8","alias_value":"6NSCU2L3","created_at":"2026-07-05T01:13:34Z"}],"graph_snapshots":[{"event_id":"sha256:80243141c5822a514cf576cfc4e34b5d24fe04f01ea6dd2146aba5078039e7b5","target":"graph","created_at":"2026-07-05T01:13:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1903.09338/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Decision trees are ubiquitous in machine learning for their ease of use and interpretability. Yet, these models are not typically employed in reinforcement learning as they cannot be updated online via stochastic gradient descent. We overcome this limitation by allowing for a gradient update over the entire tree that improves sample complexity affords interpretable policy extraction. First, we include theoretical motivation on the need for policy-gradient learning by examining the properties of gradient descent over differentiable decision trees. Second, we demonstrate that our approach equals","authors_text":"Andrew Silva, Ivan Dario Jimenez Rodriguez, Matthew Gombolay, Sung-Hyun Son, Taylor Killian","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-03-22T03:19:26Z","title":"Optimization Methods for Interpretable Differentiable Decision Trees in Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1903.09338","kind":"arxiv","version":5},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c56edcdf66e1d3b0dd387badc63713eb0aa1b0d7f0ef5bb4c3c20ed43b1931e5","target":"record","created_at":"2026-07-05T01:13:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"a538e72c701d9aab204bcdffb6da29dd8355a9dd777b140f6110a429ac0f4294","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-03-22T03:19:26Z","title_canon_sha256":"d863f4c87a8c52d4a13e6d585b6bf6bcb607db21f42e19dcd3b12d209534339d"},"schema_version":"1.0","source":{"id":"1903.09338","kind":"arxiv","version":5}},"canonical_sha256":"f3642a697bbe94d403e347236a027ddbbaaa021b312d160d3b758b5e55fde811","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f3642a697bbe94d403e347236a027ddbbaaa021b312d160d3b758b5e55fde811","first_computed_at":"2026-07-05T01:13:34.641648Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T01:13:34.641648Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"zO8Pji5BrfbS3GpXaBthY/SoSz8x1C9SYLye8lmJ7YIjwoG4Vd/SSqDr5xByWFdsXZasFeru8yty8xYv565nCw==","signature_status":"signed_v1","signed_at":"2026-07-05T01:13:34.642064Z","signed_message":"canonical_sha256_bytes"},"source_id":"1903.09338","source_kind":"arxiv","source_version":5}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c56edcdf66e1d3b0dd387badc63713eb0aa1b0d7f0ef5bb4c3c20ed43b1931e5","sha256:80243141c5822a514cf576cfc4e34b5d24fe04f01ea6dd2146aba5078039e7b5"],"state_sha256":"9749697ab3b7b88bd9958433f40c59d1405f922178ec0c64b1812d03a6d2aabe"}