{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:I4YU5LSXFIBCZCP4BKITAVSHSY","short_pith_number":"pith:I4YU5LSX","schema_version":"1.0","canonical_sha256":"47314eae572a022c89fc0a913056479613fda881c3ff5d46ec9cba13303be5f9","source":{"kind":"arxiv","id":"2501.03902","version":1},"attestation_state":"computed","paper":{"title":"Explainable Reinforcement Learning via Temporal Policy Decomposition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alessio Russo, Franco Ruggeri, Karl Henrik Johansson, Rafia Inam","submitted_at":"2025-01-07T16:10:09Z","abstract_excerpt":"We investigate the explainability of Reinforcement Learning (RL) policies from a temporal perspective, focusing on the sequence of future outcomes associated with individual actions. In RL, value functions compress information about rewards collected across multiple trajectories and over an infinite horizon, allowing a compact form of knowledge representation. However, this compression obscures the temporal details inherent in sequential decision-making, presenting a key challenge for interpretability. We present Temporal Policy Decomposition (TPD), a novel explainability approach that explain"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.03902","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-07T16:10:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e277036bf882cde12e2961880954c05b9b3adb3d8660d91675bdc85f8b9e92d4","abstract_canon_sha256":"064b08e776667f1bcf7ea6de2f934d24b6f153973d6b768f9997da7ee3cd455b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:58:03.190429Z","signature_b64":"OlP4HrQxCl9L7scSyKNp7OPK8w9fDeNzQRxjvAKiWlVn4UMuO6FZRdWgp0baQssiGdtstNK6Ije/ofHjNk9RDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"47314eae572a022c89fc0a913056479613fda881c3ff5d46ec9cba13303be5f9","last_reissued_at":"2026-07-05T09:58:03.189943Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:58:03.189943Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Explainable Reinforcement Learning via Temporal Policy Decomposition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alessio Russo, Franco Ruggeri, Karl Henrik Johansson, Rafia Inam","submitted_at":"2025-01-07T16:10:09Z","abstract_excerpt":"We investigate the explainability of Reinforcement Learning (RL) policies from a temporal perspective, focusing on the sequence of future outcomes associated with individual actions. In RL, value functions compress information about rewards collected across multiple trajectories and over an infinite horizon, allowing a compact form of knowledge representation. However, this compression obscures the temporal details inherent in sequential decision-making, presenting a key challenge for interpretability. We present Temporal Policy Decomposition (TPD), a novel explainability approach that explain"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.03902","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.03902/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.03902","created_at":"2026-07-05T09:58:03.190002+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.03902v1","created_at":"2026-07-05T09:58:03.190002+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.03902","created_at":"2026-07-05T09:58:03.190002+00:00"},{"alias_kind":"pith_short_12","alias_value":"I4YU5LSXFIBC","created_at":"2026-07-05T09:58:03.190002+00:00"},{"alias_kind":"pith_short_16","alias_value":"I4YU5LSXFIBCZCP4","created_at":"2026-07-05T09:58:03.190002+00:00"},{"alias_kind":"pith_short_8","alias_value":"I4YU5LSX","created_at":"2026-07-05T09:58:03.190002+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.03469","citing_title":"Verification-Guided Falsification for Safe RL via Explainable Abstraction and Risk-Aware Exploration","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I4YU5LSXFIBCZCP4BKITAVSHSY","json":"https://pith.science/pith/I4YU5LSXFIBCZCP4BKITAVSHSY.json","graph_json":"https://pith.science/api/pith-number/I4YU5LSXFIBCZCP4BKITAVSHSY/graph.json","events_json":"https://pith.science/api/pith-number/I4YU5LSXFIBCZCP4BKITAVSHSY/events.json","paper":"https://pith.science/paper/I4YU5LSX"},"agent_actions":{"view_html":"https://pith.science/pith/I4YU5LSXFIBCZCP4BKITAVSHSY","download_json":"https://pith.science/pith/I4YU5LSXFIBCZCP4BKITAVSHSY.json","view_paper":"https://pith.science/paper/I4YU5LSX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.03902&json=true","fetch_graph":"https://pith.science/api/pith-number/I4YU5LSXFIBCZCP4BKITAVSHSY/graph.json","fetch_events":"https://pith.science/api/pith-number/I4YU5LSXFIBCZCP4BKITAVSHSY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I4YU5LSXFIBCZCP4BKITAVSHSY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I4YU5LSXFIBCZCP4BKITAVSHSY/action/storage_attestation","attest_author":"https://pith.science/pith/I4YU5LSXFIBCZCP4BKITAVSHSY/action/author_attestation","sign_citation":"https://pith.science/pith/I4YU5LSXFIBCZCP4BKITAVSHSY/action/citation_signature","submit_replication":"https://pith.science/pith/I4YU5LSXFIBCZCP4BKITAVSHSY/action/replication_record"}},"created_at":"2026-07-05T09:58:03.190002+00:00","updated_at":"2026-07-05T09:58:03.190002+00:00"}