{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5TAJ7GY3BOYN4WRUY4QPYIGCBQ","short_pith_number":"pith:5TAJ7GY3","schema_version":"1.0","canonical_sha256":"ecc09f9b1b0bb0de5a34c720fc20c20c2d82af79f05fc73e6bf4ce524e7ef4e3","source":{"kind":"arxiv","id":"2411.15036","version":1},"attestation_state":"computed","paper":{"title":"Safe Multi-Agent Reinforcement Learning with Convergence to Generalized Nash Equilibrium","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Navid Azizan, Zeyang Li","submitted_at":"2024-11-22T16:08:42Z","abstract_excerpt":"Multi-agent reinforcement learning (MARL) has achieved notable success in cooperative tasks, demonstrating impressive performance and scalability. However, deploying MARL agents in real-world applications presents critical safety challenges. Current safe MARL algorithms are largely based on the constrained Markov decision process (CMDP) framework, which enforces constraints only on discounted cumulative costs and lacks an all-time safety assurance. Moreover, these methods often overlook the feasibility issue (the system will inevitably violate state constraints within certain regions of the co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15036","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-11-22T16:08:42Z","cross_cats_sorted":["cs.SY","eess.SY"],"title_canon_sha256":"2ed7e0d9704f3bd2aca0d03b8831da252e1801d7482f33fc2582ab0e4ea09dea","abstract_canon_sha256":"924b752a03f2c15bf143f43cf22ea2099fca6fc9f1bb0edc83234ea74b7bea7c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:15.404442Z","signature_b64":"fUvebIwRjaIFWpTUihdVYefUfiqHJ7E+UIdmvdHQdAkc2qAe3/cUgdfE/T85MIKc/CLcu2Ky7iwcvbL/3Qu1Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ecc09f9b1b0bb0de5a34c720fc20c20c2d82af79f05fc73e6bf4ce524e7ef4e3","last_reissued_at":"2026-07-05T09:39:15.403981Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:15.403981Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safe Multi-Agent Reinforcement Learning with Convergence to Generalized Nash Equilibrium","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Navid Azizan, Zeyang Li","submitted_at":"2024-11-22T16:08:42Z","abstract_excerpt":"Multi-agent reinforcement learning (MARL) has achieved notable success in cooperative tasks, demonstrating impressive performance and scalability. However, deploying MARL agents in real-world applications presents critical safety challenges. Current safe MARL algorithms are largely based on the constrained Markov decision process (CMDP) framework, which enforces constraints only on discounted cumulative costs and lacks an all-time safety assurance. Moreover, these methods often overlook the feasibility issue (the system will inevitably violate state constraints within certain regions of the co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15036","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15036/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15036","created_at":"2026-07-05T09:39:15.404052+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15036v1","created_at":"2026-07-05T09:39:15.404052+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15036","created_at":"2026-07-05T09:39:15.404052+00:00"},{"alias_kind":"pith_short_12","alias_value":"5TAJ7GY3BOYN","created_at":"2026-07-05T09:39:15.404052+00:00"},{"alias_kind":"pith_short_16","alias_value":"5TAJ7GY3BOYN4WRU","created_at":"2026-07-05T09:39:15.404052+00:00"},{"alias_kind":"pith_short_8","alias_value":"5TAJ7GY3","created_at":"2026-07-05T09:39:15.404052+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18308","citing_title":"TRIDENT: Breaking the Hybrid-Safety-Physics Coupling for Provably Safe Multi-Agent Reinforcement Learning","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03847","citing_title":"Mechanical Conscience: A Mathematical Framework for Dependability of Machine Intelligence","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03847","citing_title":"Mechanical Conscience: A Mathematical Framework for Dependability of Machine Intelligence","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14135","citing_title":"AdaFair-MARL: Enforcing Adaptive Fairness Constraints in Multi-Agent Reinforcement Learning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03847","citing_title":"Mechanical Conscience: A Mathematical Framework for Dependability of Machine Intelligence","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5TAJ7GY3BOYN4WRUY4QPYIGCBQ","json":"https://pith.science/pith/5TAJ7GY3BOYN4WRUY4QPYIGCBQ.json","graph_json":"https://pith.science/api/pith-number/5TAJ7GY3BOYN4WRUY4QPYIGCBQ/graph.json","events_json":"https://pith.science/api/pith-number/5TAJ7GY3BOYN4WRUY4QPYIGCBQ/events.json","paper":"https://pith.science/paper/5TAJ7GY3"},"agent_actions":{"view_html":"https://pith.science/pith/5TAJ7GY3BOYN4WRUY4QPYIGCBQ","download_json":"https://pith.science/pith/5TAJ7GY3BOYN4WRUY4QPYIGCBQ.json","view_paper":"https://pith.science/paper/5TAJ7GY3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15036&json=true","fetch_graph":"https://pith.science/api/pith-number/5TAJ7GY3BOYN4WRUY4QPYIGCBQ/graph.json","fetch_events":"https://pith.science/api/pith-number/5TAJ7GY3BOYN4WRUY4QPYIGCBQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5TAJ7GY3BOYN4WRUY4QPYIGCBQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5TAJ7GY3BOYN4WRUY4QPYIGCBQ/action/storage_attestation","attest_author":"https://pith.science/pith/5TAJ7GY3BOYN4WRUY4QPYIGCBQ/action/author_attestation","sign_citation":"https://pith.science/pith/5TAJ7GY3BOYN4WRUY4QPYIGCBQ/action/citation_signature","submit_replication":"https://pith.science/pith/5TAJ7GY3BOYN4WRUY4QPYIGCBQ/action/replication_record"}},"created_at":"2026-07-05T09:39:15.404052+00:00","updated_at":"2026-07-05T09:39:15.404052+00:00"}