{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:SQ5HNKKAYNJKPYP2SH5FKPVOMA","short_pith_number":"pith:SQ5HNKKA","canonical_record":{"source":{"id":"2207.00046","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-30T18:26:03Z","cross_cats_sorted":["cs.GT"],"title_canon_sha256":"cb4ce69ed320a5aef8e6bd6c9bcdcc87c318c0464de70ae67cc39fc6ba218292","abstract_canon_sha256":"c13aa06beb34d1ceddf82da1ec73e874b0b9e11ef53daac581f7ebb8ca784ed6"},"schema_version":"1.0"},"canonical_sha256":"943a76a940c352a7e1fa91fa553eae60071bf231dfb776ad8513483bcb3524f2","source":{"kind":"arxiv","id":"2207.00046","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2207.00046","created_at":"2026-07-05T06:18:16Z"},{"alias_kind":"arxiv_version","alias_value":"2207.00046v2","created_at":"2026-07-05T06:18:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.00046","created_at":"2026-07-05T06:18:16Z"},{"alias_kind":"pith_short_12","alias_value":"SQ5HNKKAYNJK","created_at":"2026-07-05T06:18:16Z"},{"alias_kind":"pith_short_16","alias_value":"SQ5HNKKAYNJKPYP2","created_at":"2026-07-05T06:18:16Z"},{"alias_kind":"pith_short_8","alias_value":"SQ5HNKKA","created_at":"2026-07-05T06:18:16Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:SQ5HNKKAYNJKPYP2SH5FKPVOMA","target":"record","payload":{"canonical_record":{"source":{"id":"2207.00046","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-30T18:26:03Z","cross_cats_sorted":["cs.GT"],"title_canon_sha256":"cb4ce69ed320a5aef8e6bd6c9bcdcc87c318c0464de70ae67cc39fc6ba218292","abstract_canon_sha256":"c13aa06beb34d1ceddf82da1ec73e874b0b9e11ef53daac581f7ebb8ca784ed6"},"schema_version":"1.0"},"canonical_sha256":"943a76a940c352a7e1fa91fa553eae60071bf231dfb776ad8513483bcb3524f2","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:18:16.961841Z","signature_b64":"MfPgiFa94Fbhtt5rQyYtfvLPXyflEkWmWYpexcyRw4c1hUUnIOUi1rgO8sj4DTQgM51SH6AoaN4aqoEwuBtUBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"943a76a940c352a7e1fa91fa553eae60071bf231dfb776ad8513483bcb3524f2","last_reissued_at":"2026-07-05T06:18:16.961433Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:18:16.961433Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2207.00046","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:18:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"QIKdyw9kX3OgYfWMjfvEngK9zjdOka/aT8DXcWFkKAJVFgEuyQyPGa2biE8CAwASqu8eXDec6xNxWAXyqXAgBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T03:45:16.066140Z"},"content_sha256":"50dc3d2828990235c88d39098ba937b8552ee266b167da20a1db518ef9b10f2f","schema_version":"1.0","event_id":"sha256:50dc3d2828990235c88d39098ba937b8552ee266b167da20a1db518ef9b10f2f"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:SQ5HNKKAYNJKPYP2SH5FKPVOMA","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Performative Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GT"],"primary_cat":"cs.LG","authors_text":"Debmalya Mandal, Goran Radanovic, Stelios Triantafyllou","submitted_at":"2022-06-30T18:26:03Z","abstract_excerpt":"We introduce the framework of performative reinforcement learning where the policy chosen by the learner affects the underlying reward and transition dynamics of the environment. Following the recent literature on performative prediction~\\cite{Perdomo et. al., 2020}, we introduce the concept of performatively stable policy. We then consider a regularized version of the reinforcement learning problem and show that repeatedly optimizing this objective converges to a performatively stable policy under reasonable assumptions on the transition dynamics. Our proof utilizes the dual perspective of th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.00046","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.00046/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:18:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Jwf3lWlYwyX3BGUOq3t7MnRMqgKgKcVwS8/ij3Xgjlv0V3Pb4QfFA1Z7afYsI36sjoRaHeNgdgUxbhOgc1tbCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T03:45:16.066931Z"},"content_sha256":"d2af1a9f9a39ee07cf5258eae870780a69c356387cfb6b3f7b50bc62bfcaa5f6","schema_version":"1.0","event_id":"sha256:d2af1a9f9a39ee07cf5258eae870780a69c356387cfb6b3f7b50bc62bfcaa5f6"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/SQ5HNKKAYNJKPYP2SH5FKPVOMA/bundle.json","state_url":"https://pith.science/pith/SQ5HNKKAYNJKPYP2SH5FKPVOMA/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/SQ5HNKKAYNJKPYP2SH5FKPVOMA/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T03:45:16Z","links":{"resolver":"https://pith.science/pith/SQ5HNKKAYNJKPYP2SH5FKPVOMA","bundle":"https://pith.science/pith/SQ5HNKKAYNJKPYP2SH5FKPVOMA/bundle.json","state":"https://pith.science/pith/SQ5HNKKAYNJKPYP2SH5FKPVOMA/state.json","well_known_bundle":"https://pith.science/.well-known/pith/SQ5HNKKAYNJKPYP2SH5FKPVOMA/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:SQ5HNKKAYNJKPYP2SH5FKPVOMA","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c13aa06beb34d1ceddf82da1ec73e874b0b9e11ef53daac581f7ebb8ca784ed6","cross_cats_sorted":["cs.GT"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-30T18:26:03Z","title_canon_sha256":"cb4ce69ed320a5aef8e6bd6c9bcdcc87c318c0464de70ae67cc39fc6ba218292"},"schema_version":"1.0","source":{"id":"2207.00046","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2207.00046","created_at":"2026-07-05T06:18:16Z"},{"alias_kind":"arxiv_version","alias_value":"2207.00046v2","created_at":"2026-07-05T06:18:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.00046","created_at":"2026-07-05T06:18:16Z"},{"alias_kind":"pith_short_12","alias_value":"SQ5HNKKAYNJK","created_at":"2026-07-05T06:18:16Z"},{"alias_kind":"pith_short_16","alias_value":"SQ5HNKKAYNJKPYP2","created_at":"2026-07-05T06:18:16Z"},{"alias_kind":"pith_short_8","alias_value":"SQ5HNKKA","created_at":"2026-07-05T06:18:16Z"}],"graph_snapshots":[{"event_id":"sha256:d2af1a9f9a39ee07cf5258eae870780a69c356387cfb6b3f7b50bc62bfcaa5f6","target":"graph","created_at":"2026-07-05T06:18:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2207.00046/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We introduce the framework of performative reinforcement learning where the policy chosen by the learner affects the underlying reward and transition dynamics of the environment. Following the recent literature on performative prediction~\\cite{Perdomo et. al., 2020}, we introduce the concept of performatively stable policy. We then consider a regularized version of the reinforcement learning problem and show that repeatedly optimizing this objective converges to a performatively stable policy under reasonable assumptions on the transition dynamics. Our proof utilizes the dual perspective of th","authors_text":"Debmalya Mandal, Goran Radanovic, Stelios Triantafyllou","cross_cats":["cs.GT"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-30T18:26:03Z","title":"Performative Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.00046","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:50dc3d2828990235c88d39098ba937b8552ee266b167da20a1db518ef9b10f2f","target":"record","created_at":"2026-07-05T06:18:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c13aa06beb34d1ceddf82da1ec73e874b0b9e11ef53daac581f7ebb8ca784ed6","cross_cats_sorted":["cs.GT"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-30T18:26:03Z","title_canon_sha256":"cb4ce69ed320a5aef8e6bd6c9bcdcc87c318c0464de70ae67cc39fc6ba218292"},"schema_version":"1.0","source":{"id":"2207.00046","kind":"arxiv","version":2}},"canonical_sha256":"943a76a940c352a7e1fa91fa553eae60071bf231dfb776ad8513483bcb3524f2","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"943a76a940c352a7e1fa91fa553eae60071bf231dfb776ad8513483bcb3524f2","first_computed_at":"2026-07-05T06:18:16.961433Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:18:16.961433Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"MfPgiFa94Fbhtt5rQyYtfvLPXyflEkWmWYpexcyRw4c1hUUnIOUi1rgO8sj4DTQgM51SH6AoaN4aqoEwuBtUBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T06:18:16.961841Z","signed_message":"canonical_sha256_bytes"},"source_id":"2207.00046","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:50dc3d2828990235c88d39098ba937b8552ee266b167da20a1db518ef9b10f2f","sha256:d2af1a9f9a39ee07cf5258eae870780a69c356387cfb6b3f7b50bc62bfcaa5f6"],"state_sha256":"d9b905d945f6564f75741c208ecce776388cbdee4d3dc2c73fe801ea650f727b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8pvUaCqZmrz5cUZzQQ7fSHxmNWAGzS4LCX9fDpMfUno80sofWZvz5CVoNYEw7tUQfdfQjLYd1ZXEFUjeMGKLBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T03:45:16.072104Z","bundle_sha256":"f9cc1fea1e8f29c465a2f329a23d0d5bf7f8db1d4f68b2668f487e2ed2fc7694"}}