{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:3X2LP3LPKDQY2OS55OYSMZNVHW","short_pith_number":"pith:3X2LP3LP","schema_version":"1.0","canonical_sha256":"ddf4b7ed6f50e18d3a5debb12665b53da9dc502324fce4aa7fb5d78ae62eba3f","source":{"kind":"arxiv","id":"2101.11196","version":2},"attestation_state":"computed","paper":{"title":"Safe Multi-Agent Reinforcement Learning via Shielding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.FL"],"primary_cat":"cs.LG","authors_text":"Christopher Amato, Ingy Elsayed-Aly, Lu Feng, R\\\"udiger Ehlers, Suda Bharadwaj, Ufuk Topcu","submitted_at":"2021-01-27T04:27:06Z","abstract_excerpt":"Multi-agent reinforcement learning (MARL) has been increasingly used in a wide range of safety-critical applications, which require guaranteed safety (e.g., no unsafe states are ever visited) during the learning process.Unfortunately, current MARL methods do not have safety guarantees. Therefore, we present two shielding approaches for safe MARL. In centralized shielding, we synthesize a single shield to monitor all agents' joint actions and correct any unsafe action if necessary. In factored shielding, we synthesize multiple shields based on a factorization of the joint state space observed b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.11196","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-27T04:27:06Z","cross_cats_sorted":["cs.FL"],"title_canon_sha256":"384be0531d394a6048e9a37a680ec6bb41a4bfe7932be8868c2a9c1e14711a10","abstract_canon_sha256":"3da80ddb0cc285821fae8ba9987fc808d852c782f2f4115ec656c30a61d6631e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:11:26.293775Z","signature_b64":"q8gx3bVR6sQdhc1MPeBtucKISapZu0t3JYH4wn2DbFFhGDtUVT1MZ8eR7D1165hRROUxyiPsmh3bdjgYJjB3BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ddf4b7ed6f50e18d3a5debb12665b53da9dc502324fce4aa7fb5d78ae62eba3f","last_reissued_at":"2026-07-05T02:11:26.293274Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:11:26.293274Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safe Multi-Agent Reinforcement Learning via Shielding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.FL"],"primary_cat":"cs.LG","authors_text":"Christopher Amato, Ingy Elsayed-Aly, Lu Feng, R\\\"udiger Ehlers, Suda Bharadwaj, Ufuk Topcu","submitted_at":"2021-01-27T04:27:06Z","abstract_excerpt":"Multi-agent reinforcement learning (MARL) has been increasingly used in a wide range of safety-critical applications, which require guaranteed safety (e.g., no unsafe states are ever visited) during the learning process.Unfortunately, current MARL methods do not have safety guarantees. Therefore, we present two shielding approaches for safe MARL. In centralized shielding, we synthesize a single shield to monitor all agents' joint actions and correct any unsafe action if necessary. In factored shielding, we synthesize multiple shields based on a factorization of the joint state space observed b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.11196","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.11196/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.11196","created_at":"2026-07-05T02:11:26.293343+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.11196v2","created_at":"2026-07-05T02:11:26.293343+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.11196","created_at":"2026-07-05T02:11:26.293343+00:00"},{"alias_kind":"pith_short_12","alias_value":"3X2LP3LPKDQY","created_at":"2026-07-05T02:11:26.293343+00:00"},{"alias_kind":"pith_short_16","alias_value":"3X2LP3LPKDQY2OS5","created_at":"2026-07-05T02:11:26.293343+00:00"},{"alias_kind":"pith_short_8","alias_value":"3X2LP3LP","created_at":"2026-07-05T02:11:26.293343+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.09128","citing_title":"A Review On Safe Reinforcement Learning Using Lyapunov and Barrier Functions","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06873","citing_title":"Generating Local Shields for Decentralised Partially Observable Markov Decision Processes","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3X2LP3LPKDQY2OS55OYSMZNVHW","json":"https://pith.science/pith/3X2LP3LPKDQY2OS55OYSMZNVHW.json","graph_json":"https://pith.science/api/pith-number/3X2LP3LPKDQY2OS55OYSMZNVHW/graph.json","events_json":"https://pith.science/api/pith-number/3X2LP3LPKDQY2OS55OYSMZNVHW/events.json","paper":"https://pith.science/paper/3X2LP3LP"},"agent_actions":{"view_html":"https://pith.science/pith/3X2LP3LPKDQY2OS55OYSMZNVHW","download_json":"https://pith.science/pith/3X2LP3LPKDQY2OS55OYSMZNVHW.json","view_paper":"https://pith.science/paper/3X2LP3LP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.11196&json=true","fetch_graph":"https://pith.science/api/pith-number/3X2LP3LPKDQY2OS55OYSMZNVHW/graph.json","fetch_events":"https://pith.science/api/pith-number/3X2LP3LPKDQY2OS55OYSMZNVHW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3X2LP3LPKDQY2OS55OYSMZNVHW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3X2LP3LPKDQY2OS55OYSMZNVHW/action/storage_attestation","attest_author":"https://pith.science/pith/3X2LP3LPKDQY2OS55OYSMZNVHW/action/author_attestation","sign_citation":"https://pith.science/pith/3X2LP3LPKDQY2OS55OYSMZNVHW/action/citation_signature","submit_replication":"https://pith.science/pith/3X2LP3LPKDQY2OS55OYSMZNVHW/action/replication_record"}},"created_at":"2026-07-05T02:11:26.293343+00:00","updated_at":"2026-07-05T02:11:26.293343+00:00"}