{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3UM7BWU4FCHVEOSSZV6BONRA7Y","short_pith_number":"pith:3UM7BWU4","schema_version":"1.0","canonical_sha256":"dd19f0da9c288f523a52cd7c173620fe0cabeecc854eeee9ad520c2c47c8c36f","source":{"kind":"arxiv","id":"2503.23101","version":2},"attestation_state":"computed","paper":{"title":"RL2Grid: Benchmarking Reinforcement Learning in Power Grid Operations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Antoine Marot, Benjamin Donnot, Cathy Wu, Christian Merz, Constance Crozier, Enrico Marchesini, Ian Dytham, Lars Schewe, Nico Westerbeck, Priya L. Donti","submitted_at":"2025-03-29T14:39:17Z","abstract_excerpt":"Reinforcement learning (RL) can provide adaptive and scalable controllers essential for power grid decarbonization. However, RL methods struggle with power grids' complex dynamics, long-horizon goals, and hard physical constraints. For these reasons, we present RL2Grid, a benchmark designed in collaboration with power system operators to accelerate progress in grid control and foster RL maturity. Built on RTE France's power simulation framework, RL2Grid standardizes tasks, state and action spaces, and reward structures for a systematic evaluation and comparison of RL algorithms. Moreover, we i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.23101","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-29T14:39:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b3d477a918ae315d26275a8c330c3d0f6af49e72bfd76caf5f92eb1bc3004a7a","abstract_canon_sha256":"af640445e66a9b712ee95527abd371ad7a1fbc7d7f8d3e3c60c235dbb35cd523"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:24:17.583265Z","signature_b64":"/xq8mYovM/CYGoUgoD7eElfH1/8/C5bW6u4Ixtjikp/7WYv6vKlFKrds9OwLdBSFy5LcKs1DPoPypFhT0b7DCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd19f0da9c288f523a52cd7c173620fe0cabeecc854eeee9ad520c2c47c8c36f","last_reissued_at":"2026-07-05T11:24:17.582744Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:24:17.582744Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RL2Grid: Benchmarking Reinforcement Learning in Power Grid Operations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Antoine Marot, Benjamin Donnot, Cathy Wu, Christian Merz, Constance Crozier, Enrico Marchesini, Ian Dytham, Lars Schewe, Nico Westerbeck, Priya L. Donti","submitted_at":"2025-03-29T14:39:17Z","abstract_excerpt":"Reinforcement learning (RL) can provide adaptive and scalable controllers essential for power grid decarbonization. However, RL methods struggle with power grids' complex dynamics, long-horizon goals, and hard physical constraints. For these reasons, we present RL2Grid, a benchmark designed in collaboration with power system operators to accelerate progress in grid control and foster RL maturity. Built on RTE France's power simulation framework, RL2Grid standardizes tasks, state and action spaces, and reward structures for a systematic evaluation and comparison of RL algorithms. Moreover, we i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.23101","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.23101/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.23101","created_at":"2026-07-05T11:24:17.582812+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.23101v2","created_at":"2026-07-05T11:24:17.582812+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.23101","created_at":"2026-07-05T11:24:17.582812+00:00"},{"alias_kind":"pith_short_12","alias_value":"3UM7BWU4FCHV","created_at":"2026-07-05T11:24:17.582812+00:00"},{"alias_kind":"pith_short_16","alias_value":"3UM7BWU4FCHVEOSS","created_at":"2026-07-05T11:24:17.582812+00:00"},{"alias_kind":"pith_short_8","alias_value":"3UM7BWU4","created_at":"2026-07-05T11:24:17.582812+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14758","citing_title":"Probabilistic Verification of Recurrent Neural Networks for Single and Multi-Agent Reinforcement Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00561","citing_title":"Interpretable Policy Distillation for Power Grid Topology Control","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03142","citing_title":"MARS-DA: A Hierarchical Reinforcement Learning Framework for Risk-Aware Multi-Agent Bidding in Power Grids","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05519","citing_title":"OpenG2G: A Simulation Platform for AI Datacenter-Grid Runtime Coordination","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3UM7BWU4FCHVEOSSZV6BONRA7Y","json":"https://pith.science/pith/3UM7BWU4FCHVEOSSZV6BONRA7Y.json","graph_json":"https://pith.science/api/pith-number/3UM7BWU4FCHVEOSSZV6BONRA7Y/graph.json","events_json":"https://pith.science/api/pith-number/3UM7BWU4FCHVEOSSZV6BONRA7Y/events.json","paper":"https://pith.science/paper/3UM7BWU4"},"agent_actions":{"view_html":"https://pith.science/pith/3UM7BWU4FCHVEOSSZV6BONRA7Y","download_json":"https://pith.science/pith/3UM7BWU4FCHVEOSSZV6BONRA7Y.json","view_paper":"https://pith.science/paper/3UM7BWU4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.23101&json=true","fetch_graph":"https://pith.science/api/pith-number/3UM7BWU4FCHVEOSSZV6BONRA7Y/graph.json","fetch_events":"https://pith.science/api/pith-number/3UM7BWU4FCHVEOSSZV6BONRA7Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3UM7BWU4FCHVEOSSZV6BONRA7Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3UM7BWU4FCHVEOSSZV6BONRA7Y/action/storage_attestation","attest_author":"https://pith.science/pith/3UM7BWU4FCHVEOSSZV6BONRA7Y/action/author_attestation","sign_citation":"https://pith.science/pith/3UM7BWU4FCHVEOSSZV6BONRA7Y/action/citation_signature","submit_replication":"https://pith.science/pith/3UM7BWU4FCHVEOSSZV6BONRA7Y/action/replication_record"}},"created_at":"2026-07-05T11:24:17.582812+00:00","updated_at":"2026-07-05T11:24:17.582812+00:00"}