{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ODTY2TJMPYNZZ45QPQDGMLW75Z","short_pith_number":"pith:ODTY2TJM","schema_version":"1.0","canonical_sha256":"70e78d4d2c7e1b9cf3b07c06662edfee6f1baa49a3146481020dc2b6332464d6","source":{"kind":"arxiv","id":"2305.10378","version":1},"attestation_state":"computed","paper":{"title":"Explainable Multi-Agent Reinforcement Learning for Temporal Queries","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Kayla Boggess, Lu Feng, Sarit Kraus","submitted_at":"2023-05-17T17:04:29Z","abstract_excerpt":"As multi-agent reinforcement learning (MARL) systems are increasingly deployed throughout society, it is imperative yet challenging for users to understand the emergent behaviors of MARL agents in complex environments. This work presents an approach for generating policy-level contrastive explanations for MARL to answer a temporal user query, which specifies a sequence of tasks completed by agents with possible cooperation. The proposed approach encodes the temporal query as a PCTL logic formula and checks if the query is feasible under a given MARL policy via probabilistic model checking. Suc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.10378","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-05-17T17:04:29Z","cross_cats_sorted":[],"title_canon_sha256":"f08fa652ee190add81e40bdf7ad10f6467d5084dba4d6c3460ea7da597a4ba4b","abstract_canon_sha256":"6dd15148aa519b0797fcc87f1f6f380cfa0b3a2db4044f85b6f22301cd302d09"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:11:13.736751Z","signature_b64":"hQUsnG+75Q6C43utAGXcwsZrUNepNplDOHEFJsMOQ38wDCoxue2x9Z5F7PQ31stKgG/0egO6NqBp8nJbEVHNAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"70e78d4d2c7e1b9cf3b07c06662edfee6f1baa49a3146481020dc2b6332464d6","last_reissued_at":"2026-07-05T06:11:13.736307Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:11:13.736307Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Explainable Multi-Agent Reinforcement Learning for Temporal Queries","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Kayla Boggess, Lu Feng, Sarit Kraus","submitted_at":"2023-05-17T17:04:29Z","abstract_excerpt":"As multi-agent reinforcement learning (MARL) systems are increasingly deployed throughout society, it is imperative yet challenging for users to understand the emergent behaviors of MARL agents in complex environments. This work presents an approach for generating policy-level contrastive explanations for MARL to answer a temporal user query, which specifies a sequence of tasks completed by agents with possible cooperation. The proposed approach encodes the temporal query as a PCTL logic formula and checks if the query is feasible under a given MARL policy via probabilistic model checking. Suc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.10378","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.10378/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.10378","created_at":"2026-07-05T06:11:13.736385+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.10378v1","created_at":"2026-07-05T06:11:13.736385+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.10378","created_at":"2026-07-05T06:11:13.736385+00:00"},{"alias_kind":"pith_short_12","alias_value":"ODTY2TJMPYNZ","created_at":"2026-07-05T06:11:13.736385+00:00"},{"alias_kind":"pith_short_16","alias_value":"ODTY2TJMPYNZZ45Q","created_at":"2026-07-05T06:11:13.736385+00:00"},{"alias_kind":"pith_short_8","alias_value":"ODTY2TJM","created_at":"2026-07-05T06:11:13.736385+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.00684","citing_title":"Compositional Concept-Based Neuron-Level Interpretability for Deep Reinforcement Learning","ref_index":2017,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ODTY2TJMPYNZZ45QPQDGMLW75Z","json":"https://pith.science/pith/ODTY2TJMPYNZZ45QPQDGMLW75Z.json","graph_json":"https://pith.science/api/pith-number/ODTY2TJMPYNZZ45QPQDGMLW75Z/graph.json","events_json":"https://pith.science/api/pith-number/ODTY2TJMPYNZZ45QPQDGMLW75Z/events.json","paper":"https://pith.science/paper/ODTY2TJM"},"agent_actions":{"view_html":"https://pith.science/pith/ODTY2TJMPYNZZ45QPQDGMLW75Z","download_json":"https://pith.science/pith/ODTY2TJMPYNZZ45QPQDGMLW75Z.json","view_paper":"https://pith.science/paper/ODTY2TJM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.10378&json=true","fetch_graph":"https://pith.science/api/pith-number/ODTY2TJMPYNZZ45QPQDGMLW75Z/graph.json","fetch_events":"https://pith.science/api/pith-number/ODTY2TJMPYNZZ45QPQDGMLW75Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ODTY2TJMPYNZZ45QPQDGMLW75Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ODTY2TJMPYNZZ45QPQDGMLW75Z/action/storage_attestation","attest_author":"https://pith.science/pith/ODTY2TJMPYNZZ45QPQDGMLW75Z/action/author_attestation","sign_citation":"https://pith.science/pith/ODTY2TJMPYNZZ45QPQDGMLW75Z/action/citation_signature","submit_replication":"https://pith.science/pith/ODTY2TJMPYNZZ45QPQDGMLW75Z/action/replication_record"}},"created_at":"2026-07-05T06:11:13.736385+00:00","updated_at":"2026-07-05T06:11:13.736385+00:00"}