{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:Q35J6J5NG5YX2QF2BXPROXFKBF","short_pith_number":"pith:Q35J6J5N","schema_version":"1.0","canonical_sha256":"86fa9f27ad37717d40ba0ddf175caa09689aa93355ca48e35bb5aa03652dd555","source":{"kind":"arxiv","id":"2110.02793","version":2},"attestation_state":"computed","paper":{"title":"Multi-Agent Constrained Policy Optimisation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.AI","authors_text":"Alois Knoll, Jakub Grudzien Kuba, Jun Wang, Munning Wen, Ruiqing Chen, Shangding Gu, Yaodong Yang, Zheng Tian, Ziyan Wang","submitted_at":"2021-10-06T14:17:09Z","abstract_excerpt":"Developing reinforcement learning algorithms that satisfy safety constraints is becoming increasingly important in real-world applications. In multi-agent reinforcement learning (MARL) settings, policy optimisation with safety awareness is particularly challenging because each individual agent has to not only meet its own safety constraints, but also consider those of others so that their joint behaviour can be guaranteed safe. Despite its importance, the problem of safe multi-agent learning has not been rigorously studied; very few solutions have been proposed, nor a sharable testing environm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.02793","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2021-10-06T14:17:09Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"1b030f305757c127ccba6d04cf528bfff1e50032840345e86bbdbb90b6f43aae","abstract_canon_sha256":"d7787f9715a640e163a3d868be8a95be87d61908812590ae411b959533e232a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:55:49.970636Z","signature_b64":"5yFrxJg8nk0R9rGfOc675MazZzzmu0Shu6jnArs9+k0jQtvVeGrZhlLFLDcLpxKt7yJK70TjKnWgAC0POhgUCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86fa9f27ad37717d40ba0ddf175caa09689aa93355ca48e35bb5aa03652dd555","last_reissued_at":"2026-07-05T03:55:49.970158Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:55:49.970158Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Agent Constrained Policy Optimisation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.AI","authors_text":"Alois Knoll, Jakub Grudzien Kuba, Jun Wang, Munning Wen, Ruiqing Chen, Shangding Gu, Yaodong Yang, Zheng Tian, Ziyan Wang","submitted_at":"2021-10-06T14:17:09Z","abstract_excerpt":"Developing reinforcement learning algorithms that satisfy safety constraints is becoming increasingly important in real-world applications. In multi-agent reinforcement learning (MARL) settings, policy optimisation with safety awareness is particularly challenging because each individual agent has to not only meet its own safety constraints, but also consider those of others so that their joint behaviour can be guaranteed safe. Despite its importance, the problem of safe multi-agent learning has not been rigorously studied; very few solutions have been proposed, nor a sharable testing environm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.02793","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.02793/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.02793","created_at":"2026-07-05T03:55:49.970210+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.02793v2","created_at":"2026-07-05T03:55:49.970210+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.02793","created_at":"2026-07-05T03:55:49.970210+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q35J6J5NG5YX","created_at":"2026-07-05T03:55:49.970210+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q35J6J5NG5YX2QF2","created_at":"2026-07-05T03:55:49.970210+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q35J6J5N","created_at":"2026-07-05T03:55:49.970210+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25526","citing_title":"Low Variance Trust Region Optimization with Independent Actors and Sequential Updates in Cooperative Multi-agent Reinforcement Learning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24010","citing_title":"Safe and Generalizable Hierarchical Multi-Agent RL via Constraint Manifold Control","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18308","citing_title":"TRIDENT: Breaking the Hybrid-Safety-Physics Coupling for Provably Safe Multi-Agent Reinforcement Learning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11284","citing_title":"Phi-Actor-Critic: Steering General-Sum Games to Pareto-Efficient Correlated Equilibria","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q35J6J5NG5YX2QF2BXPROXFKBF","json":"https://pith.science/pith/Q35J6J5NG5YX2QF2BXPROXFKBF.json","graph_json":"https://pith.science/api/pith-number/Q35J6J5NG5YX2QF2BXPROXFKBF/graph.json","events_json":"https://pith.science/api/pith-number/Q35J6J5NG5YX2QF2BXPROXFKBF/events.json","paper":"https://pith.science/paper/Q35J6J5N"},"agent_actions":{"view_html":"https://pith.science/pith/Q35J6J5NG5YX2QF2BXPROXFKBF","download_json":"https://pith.science/pith/Q35J6J5NG5YX2QF2BXPROXFKBF.json","view_paper":"https://pith.science/paper/Q35J6J5N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.02793&json=true","fetch_graph":"https://pith.science/api/pith-number/Q35J6J5NG5YX2QF2BXPROXFKBF/graph.json","fetch_events":"https://pith.science/api/pith-number/Q35J6J5NG5YX2QF2BXPROXFKBF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q35J6J5NG5YX2QF2BXPROXFKBF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q35J6J5NG5YX2QF2BXPROXFKBF/action/storage_attestation","attest_author":"https://pith.science/pith/Q35J6J5NG5YX2QF2BXPROXFKBF/action/author_attestation","sign_citation":"https://pith.science/pith/Q35J6J5NG5YX2QF2BXPROXFKBF/action/citation_signature","submit_replication":"https://pith.science/pith/Q35J6J5NG5YX2QF2BXPROXFKBF/action/replication_record"}},"created_at":"2026-07-05T03:55:49.970210+00:00","updated_at":"2026-07-05T03:55:49.970210+00:00"}