{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:MKQCDSON4ADSZMGQQZHWPB6TZO","short_pith_number":"pith:MKQCDSON","schema_version":"1.0","canonical_sha256":"62a021c9cde0072cb0d0864f6787d3cbb38ddc768ac2c6e2e80c10f6e3ec3368","source":{"kind":"arxiv","id":"2001.11718","version":1},"attestation_state":"computed","paper":{"title":"Locally Private Distributed Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Hajime Ono, Tsubasa Takahashi","submitted_at":"2020-01-31T09:03:23Z","abstract_excerpt":"We study locally differentially private algorithms for reinforcement learning to obtain a robust policy that performs well across distributed private environments. Our algorithm protects the information of local agents' models from being exploited by adversarial reverse engineering. Since a local policy is strongly being affected by the individual environment, the output of the agent may release the private information unconsciously. In our proposed algorithm, local agents update the model in their environments and report noisy gradients designed to satisfy local differential privacy (LDP) tha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2001.11718","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-01-31T09:03:23Z","cross_cats_sorted":["cs.CR","stat.ML"],"title_canon_sha256":"970f8770ba734aa65694ffcb9bca24751d429b46235c865617a8d5b45a0f9d8e","abstract_canon_sha256":"2ec9971e13551be73bec22d1c05192722522b45bf85fdfbb47a0b68e90b2c1f9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:37:33.156279Z","signature_b64":"G2egQo5HQDcly9r37ZoHaAiqDef7tEt0IqeoTBF99yUq5Pr7Gk0AUA34jvZRgXeBXZh2I342BDIzkquLQPdLBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"62a021c9cde0072cb0d0864f6787d3cbb38ddc768ac2c6e2e80c10f6e3ec3368","last_reissued_at":"2026-07-05T00:37:33.155930Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:37:33.155930Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Locally Private Distributed Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Hajime Ono, Tsubasa Takahashi","submitted_at":"2020-01-31T09:03:23Z","abstract_excerpt":"We study locally differentially private algorithms for reinforcement learning to obtain a robust policy that performs well across distributed private environments. Our algorithm protects the information of local agents' models from being exploited by adversarial reverse engineering. Since a local policy is strongly being affected by the individual environment, the output of the agent may release the private information unconsciously. In our proposed algorithm, local agents update the model in their environments and report noisy gradients designed to satisfy local differential privacy (LDP) tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2001.11718","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2001.11718/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2001.11718","created_at":"2026-07-05T00:37:33.155994+00:00"},{"alias_kind":"arxiv_version","alias_value":"2001.11718v1","created_at":"2026-07-05T00:37:33.155994+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2001.11718","created_at":"2026-07-05T00:37:33.155994+00:00"},{"alias_kind":"pith_short_12","alias_value":"MKQCDSON4ADS","created_at":"2026-07-05T00:37:33.155994+00:00"},{"alias_kind":"pith_short_16","alias_value":"MKQCDSON4ADSZMGQ","created_at":"2026-07-05T00:37:33.155994+00:00"},{"alias_kind":"pith_short_8","alias_value":"MKQCDSON","created_at":"2026-07-05T00:37:33.155994+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.20621","citing_title":"Multi-level Certified Defense Against Poisoning Attacks in Offline Reinforcement Learning","ref_index":66,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MKQCDSON4ADSZMGQQZHWPB6TZO","json":"https://pith.science/pith/MKQCDSON4ADSZMGQQZHWPB6TZO.json","graph_json":"https://pith.science/api/pith-number/MKQCDSON4ADSZMGQQZHWPB6TZO/graph.json","events_json":"https://pith.science/api/pith-number/MKQCDSON4ADSZMGQQZHWPB6TZO/events.json","paper":"https://pith.science/paper/MKQCDSON"},"agent_actions":{"view_html":"https://pith.science/pith/MKQCDSON4ADSZMGQQZHWPB6TZO","download_json":"https://pith.science/pith/MKQCDSON4ADSZMGQQZHWPB6TZO.json","view_paper":"https://pith.science/paper/MKQCDSON","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2001.11718&json=true","fetch_graph":"https://pith.science/api/pith-number/MKQCDSON4ADSZMGQQZHWPB6TZO/graph.json","fetch_events":"https://pith.science/api/pith-number/MKQCDSON4ADSZMGQQZHWPB6TZO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MKQCDSON4ADSZMGQQZHWPB6TZO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MKQCDSON4ADSZMGQQZHWPB6TZO/action/storage_attestation","attest_author":"https://pith.science/pith/MKQCDSON4ADSZMGQQZHWPB6TZO/action/author_attestation","sign_citation":"https://pith.science/pith/MKQCDSON4ADSZMGQQZHWPB6TZO/action/citation_signature","submit_replication":"https://pith.science/pith/MKQCDSON4ADSZMGQQZHWPB6TZO/action/replication_record"}},"created_at":"2026-07-05T00:37:33.155994+00:00","updated_at":"2026-07-05T00:37:33.155994+00:00"}