{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UDS5CUAECK2Y3DGKCHCQAJU332","short_pith_number":"pith:UDS5CUAE","schema_version":"1.0","canonical_sha256":"a0e5d1500412b58d8cca11c500269bde87d02622f0f834d2990d69defc5b1a37","source":{"kind":"arxiv","id":"2512.05291","version":3},"attestation_state":"computed","paper":{"title":"SHAP-Guided Kernel Actor-Critic for Explainable Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Hangguan Shan, Na Li, Wei Ni, Wenjie Zhang, Xinyu Li","submitted_at":"2025-12-04T22:28:43Z","abstract_excerpt":"Actor-critic (AC) methods are a cornerstone of reinforcement learning (RL) but offer limited interpretability. Current explainable RL methods seldom use state attributions to assist training. Rather, they treat all state features equally, thereby neglecting the heterogeneous impacts of individual state dimensions on the reward. We propose RKHS-SHAP-based Advanced Actor-Critic (RSA2C), an attribution-aware, kernelized, two-timescale AC algorithm, including Actor, Value Critic, and Advantage Critic. The Actor is instantiated in a vector-valued reproducing kernel Hilbert space (RKHS) with a Mahal"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2512.05291","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-12-04T22:28:43Z","cross_cats_sorted":[],"title_canon_sha256":"89dd86c7a2e770b0240e42b7c107457898cca68a166929713cddf8c6b4a99d9e","abstract_canon_sha256":"fe1be019743efd3f1e73f5d1d7f15f25781f843cbe4b5847eaa95522a991aa28"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-08T01:03:52.507658Z","signature_b64":"GIyRsjxWtVLxueVpQ5dEVVm8AlAFrUJl2rYpacVI/vyVOP5Am0jhFCq+HkHgrKtu7L6TTpaqmwNY7o2oPn+/Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a0e5d1500412b58d8cca11c500269bde87d02622f0f834d2990d69defc5b1a37","last_reissued_at":"2026-06-08T01:03:52.506647Z","signature_status":"signed_v1","first_computed_at":"2026-06-08T01:03:52.506647Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SHAP-Guided Kernel Actor-Critic for Explainable Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Hangguan Shan, Na Li, Wei Ni, Wenjie Zhang, Xinyu Li","submitted_at":"2025-12-04T22:28:43Z","abstract_excerpt":"Actor-critic (AC) methods are a cornerstone of reinforcement learning (RL) but offer limited interpretability. Current explainable RL methods seldom use state attributions to assist training. Rather, they treat all state features equally, thereby neglecting the heterogeneous impacts of individual state dimensions on the reward. We propose RKHS-SHAP-based Advanced Actor-Critic (RSA2C), an attribution-aware, kernelized, two-timescale AC algorithm, including Actor, Value Critic, and Advantage Critic. The Actor is instantiated in a vector-valued reproducing kernel Hilbert space (RKHS) with a Mahal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.05291","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2512.05291/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2512.05291","created_at":"2026-06-08T01:03:52.506774+00:00"},{"alias_kind":"arxiv_version","alias_value":"2512.05291v3","created_at":"2026-06-08T01:03:52.506774+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.05291","created_at":"2026-06-08T01:03:52.506774+00:00"},{"alias_kind":"pith_short_12","alias_value":"UDS5CUAECK2Y","created_at":"2026-06-08T01:03:52.506774+00:00"},{"alias_kind":"pith_short_16","alias_value":"UDS5CUAECK2Y3DGK","created_at":"2026-06-08T01:03:52.506774+00:00"},{"alias_kind":"pith_short_8","alias_value":"UDS5CUAE","created_at":"2026-06-08T01:03:52.506774+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UDS5CUAECK2Y3DGKCHCQAJU332","json":"https://pith.science/pith/UDS5CUAECK2Y3DGKCHCQAJU332.json","graph_json":"https://pith.science/api/pith-number/UDS5CUAECK2Y3DGKCHCQAJU332/graph.json","events_json":"https://pith.science/api/pith-number/UDS5CUAECK2Y3DGKCHCQAJU332/events.json","paper":"https://pith.science/paper/UDS5CUAE"},"agent_actions":{"view_html":"https://pith.science/pith/UDS5CUAECK2Y3DGKCHCQAJU332","download_json":"https://pith.science/pith/UDS5CUAECK2Y3DGKCHCQAJU332.json","view_paper":"https://pith.science/paper/UDS5CUAE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2512.05291&json=true","fetch_graph":"https://pith.science/api/pith-number/UDS5CUAECK2Y3DGKCHCQAJU332/graph.json","fetch_events":"https://pith.science/api/pith-number/UDS5CUAECK2Y3DGKCHCQAJU332/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UDS5CUAECK2Y3DGKCHCQAJU332/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UDS5CUAECK2Y3DGKCHCQAJU332/action/storage_attestation","attest_author":"https://pith.science/pith/UDS5CUAECK2Y3DGKCHCQAJU332/action/author_attestation","sign_citation":"https://pith.science/pith/UDS5CUAECK2Y3DGKCHCQAJU332/action/citation_signature","submit_replication":"https://pith.science/pith/UDS5CUAECK2Y3DGKCHCQAJU332/action/replication_record"}},"created_at":"2026-06-08T01:03:52.506774+00:00","updated_at":"2026-06-08T01:03:52.506774+00:00"}