{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MFOONFVY7QFYYIKYWXW45FMSAD","short_pith_number":"pith:MFOONFVY","schema_version":"1.0","canonical_sha256":"615ce696b8fc0b8c2158b5edce959200e38db1113129f097e5c20abbd6686b18","source":{"kind":"arxiv","id":"2501.19080","version":1},"attestation_state":"computed","paper":{"title":"Differentially Private Policy Gradient","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexandre Rio, Igor Colin, Merwan Barlier","submitted_at":"2025-01-31T12:11:13Z","abstract_excerpt":"Motivated by the increasing deployment of reinforcement learning in the real world, involving a large consumption of personal data, we introduce a differentially private (DP) policy gradient algorithm. We show that, in this setting, the introduction of Differential Privacy can be reduced to the computation of appropriate trust regions, thus avoiding the sacrifice of theoretical properties of the DP-less methods. Therefore, we show that it is possible to find the right trade-off between privacy noise and trust-region size to obtain a performant differentially private policy gradient algorithm. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.19080","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-31T12:11:13Z","cross_cats_sorted":[],"title_canon_sha256":"ac8a5964292268603f30fd35e372679d5bb3d0f1a998f49c0c48d8a8a9b98657","abstract_canon_sha256":"1fe54e3aa719e004e957e8f6a9f05f5801343e81006c0d6d7ee23ba5f983c89a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:07:55.784121Z","signature_b64":"PSSD7VWTPWkpj/c4+Ri7Pj2QtTAE1idcFuMa8ns4W1Bwjp79f4e9J91F1EIFy1n5jJQlj2ViHa8rPtTeH5U+BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"615ce696b8fc0b8c2158b5edce959200e38db1113129f097e5c20abbd6686b18","last_reissued_at":"2026-07-05T10:07:55.783619Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:07:55.783619Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Differentially Private Policy Gradient","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexandre Rio, Igor Colin, Merwan Barlier","submitted_at":"2025-01-31T12:11:13Z","abstract_excerpt":"Motivated by the increasing deployment of reinforcement learning in the real world, involving a large consumption of personal data, we introduce a differentially private (DP) policy gradient algorithm. We show that, in this setting, the introduction of Differential Privacy can be reduced to the computation of appropriate trust regions, thus avoiding the sacrifice of theoretical properties of the DP-less methods. Therefore, we show that it is possible to find the right trade-off between privacy noise and trust-region size to obtain a performant differentially private policy gradient algorithm. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.19080","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.19080/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.19080","created_at":"2026-07-05T10:07:55.783682+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.19080v1","created_at":"2026-07-05T10:07:55.783682+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.19080","created_at":"2026-07-05T10:07:55.783682+00:00"},{"alias_kind":"pith_short_12","alias_value":"MFOONFVY7QFY","created_at":"2026-07-05T10:07:55.783682+00:00"},{"alias_kind":"pith_short_16","alias_value":"MFOONFVY7QFYYIKY","created_at":"2026-07-05T10:07:55.783682+00:00"},{"alias_kind":"pith_short_8","alias_value":"MFOONFVY","created_at":"2026-07-05T10:07:55.783682+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.21060","citing_title":"On the Sample Complexity of Differentially Private Policy Optimization","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MFOONFVY7QFYYIKYWXW45FMSAD","json":"https://pith.science/pith/MFOONFVY7QFYYIKYWXW45FMSAD.json","graph_json":"https://pith.science/api/pith-number/MFOONFVY7QFYYIKYWXW45FMSAD/graph.json","events_json":"https://pith.science/api/pith-number/MFOONFVY7QFYYIKYWXW45FMSAD/events.json","paper":"https://pith.science/paper/MFOONFVY"},"agent_actions":{"view_html":"https://pith.science/pith/MFOONFVY7QFYYIKYWXW45FMSAD","download_json":"https://pith.science/pith/MFOONFVY7QFYYIKYWXW45FMSAD.json","view_paper":"https://pith.science/paper/MFOONFVY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.19080&json=true","fetch_graph":"https://pith.science/api/pith-number/MFOONFVY7QFYYIKYWXW45FMSAD/graph.json","fetch_events":"https://pith.science/api/pith-number/MFOONFVY7QFYYIKYWXW45FMSAD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MFOONFVY7QFYYIKYWXW45FMSAD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MFOONFVY7QFYYIKYWXW45FMSAD/action/storage_attestation","attest_author":"https://pith.science/pith/MFOONFVY7QFYYIKYWXW45FMSAD/action/author_attestation","sign_citation":"https://pith.science/pith/MFOONFVY7QFYYIKYWXW45FMSAD/action/citation_signature","submit_replication":"https://pith.science/pith/MFOONFVY7QFYYIKYWXW45FMSAD/action/replication_record"}},"created_at":"2026-07-05T10:07:55.783682+00:00","updated_at":"2026-07-05T10:07:55.783682+00:00"}