{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:EO4DLFWX6HOMJWNKEWBMY4VD4P","short_pith_number":"pith:EO4DLFWX","canonical_record":{"source":{"id":"2306.09554","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-15T23:51:46Z","cross_cats_sorted":[],"title_canon_sha256":"4601499b7dec805fcb1b89a97508d7f2bbddc9bf61e3aaef75a51daf03969fe3","abstract_canon_sha256":"9eaf3046de77aaec681313b523e261602c7eead8b31ec71e00a30083af1f1307"},"schema_version":"1.0"},"canonical_sha256":"23b83596d7f1dcc4d9aa2582cc72a3e3f491ae862371a86d60c51b17bf6cae63","source":{"kind":"arxiv","id":"2306.09554","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2306.09554","created_at":"2026-07-05T06:21:21Z"},{"alias_kind":"arxiv_version","alias_value":"2306.09554v1","created_at":"2026-07-05T06:21:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.09554","created_at":"2026-07-05T06:21:21Z"},{"alias_kind":"pith_short_12","alias_value":"EO4DLFWX6HOM","created_at":"2026-07-05T06:21:21Z"},{"alias_kind":"pith_short_16","alias_value":"EO4DLFWX6HOMJWNK","created_at":"2026-07-05T06:21:21Z"},{"alias_kind":"pith_short_8","alias_value":"EO4DLFWX","created_at":"2026-07-05T06:21:21Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:EO4DLFWX6HOMJWNKEWBMY4VD4P","target":"record","payload":{"canonical_record":{"source":{"id":"2306.09554","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-15T23:51:46Z","cross_cats_sorted":[],"title_canon_sha256":"4601499b7dec805fcb1b89a97508d7f2bbddc9bf61e3aaef75a51daf03969fe3","abstract_canon_sha256":"9eaf3046de77aaec681313b523e261602c7eead8b31ec71e00a30083af1f1307"},"schema_version":"1.0"},"canonical_sha256":"23b83596d7f1dcc4d9aa2582cc72a3e3f491ae862371a86d60c51b17bf6cae63","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:21:21.097710Z","signature_b64":"rbMc0pUAjZ9RQ7uSwjCwbHo0DKMkBqCp6zenikux8iGK6CAkl/kKTb+1+73uQItj+2z94IzK2/l0mzitrOPSBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23b83596d7f1dcc4d9aa2582cc72a3e3f491ae862371a86d60c51b17bf6cae63","last_reissued_at":"2026-07-05T06:21:21.097300Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:21:21.097300Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2306.09554","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:21:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"YYDULJadv4iABrNUooZLl24Is2w/amyFdnork5qFW+Q50tc3jVsJZTh5k7LQclhyk0fO7uXfNijZXuSOvwFnAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T07:06:14.168763Z"},"content_sha256":"415a9e188531f93120103ecbbfaa1c641b506f7f44d79cc5c924abb35ec95073","schema_version":"1.0","event_id":"sha256:415a9e188531f93120103ecbbfaa1c641b506f7f44d79cc5c924abb35ec95073"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:EO4DLFWX6HOMJWNKEWBMY4VD4P","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Low-Switching Policy Gradient with Exploration via Online Sensitivity Sampling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Lin Yang, Yiran Wang, Yu Cheng, Yunfan Li","submitted_at":"2023-06-15T23:51:46Z","abstract_excerpt":"Policy optimization methods are powerful algorithms in Reinforcement Learning (RL) for their flexibility to deal with policy parameterization and ability to handle model misspecification. However, these methods usually suffer from slow convergence rates and poor sample complexity. Hence it is important to design provably sample efficient algorithms for policy optimization. Yet, recent advances for this problems have only been successful in tabular and linear setting, whose benign structures cannot be generalized to non-linearly parameterized policies. In this paper, we address this problem by "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.09554","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.09554/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:21:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0ShHbJe938b5SAbjuzxRjCNk+Desw3OOVgytoFSOs1yskwJ9uYOtwVDAvMgjc9rt6M48t/ZjJKFdzNj+JPnnCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T07:06:14.169241Z"},"content_sha256":"16330f7a13429323737e0f60fb771184f8be2af1a0e8b3a442177dab07bfd049","schema_version":"1.0","event_id":"sha256:16330f7a13429323737e0f60fb771184f8be2af1a0e8b3a442177dab07bfd049"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/EO4DLFWX6HOMJWNKEWBMY4VD4P/bundle.json","state_url":"https://pith.science/pith/EO4DLFWX6HOMJWNKEWBMY4VD4P/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/EO4DLFWX6HOMJWNKEWBMY4VD4P/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T07:06:14Z","links":{"resolver":"https://pith.science/pith/EO4DLFWX6HOMJWNKEWBMY4VD4P","bundle":"https://pith.science/pith/EO4DLFWX6HOMJWNKEWBMY4VD4P/bundle.json","state":"https://pith.science/pith/EO4DLFWX6HOMJWNKEWBMY4VD4P/state.json","well_known_bundle":"https://pith.science/.well-known/pith/EO4DLFWX6HOMJWNKEWBMY4VD4P/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:EO4DLFWX6HOMJWNKEWBMY4VD4P","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9eaf3046de77aaec681313b523e261602c7eead8b31ec71e00a30083af1f1307","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-15T23:51:46Z","title_canon_sha256":"4601499b7dec805fcb1b89a97508d7f2bbddc9bf61e3aaef75a51daf03969fe3"},"schema_version":"1.0","source":{"id":"2306.09554","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2306.09554","created_at":"2026-07-05T06:21:21Z"},{"alias_kind":"arxiv_version","alias_value":"2306.09554v1","created_at":"2026-07-05T06:21:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.09554","created_at":"2026-07-05T06:21:21Z"},{"alias_kind":"pith_short_12","alias_value":"EO4DLFWX6HOM","created_at":"2026-07-05T06:21:21Z"},{"alias_kind":"pith_short_16","alias_value":"EO4DLFWX6HOMJWNK","created_at":"2026-07-05T06:21:21Z"},{"alias_kind":"pith_short_8","alias_value":"EO4DLFWX","created_at":"2026-07-05T06:21:21Z"}],"graph_snapshots":[{"event_id":"sha256:16330f7a13429323737e0f60fb771184f8be2af1a0e8b3a442177dab07bfd049","target":"graph","created_at":"2026-07-05T06:21:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2306.09554/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Policy optimization methods are powerful algorithms in Reinforcement Learning (RL) for their flexibility to deal with policy parameterization and ability to handle model misspecification. However, these methods usually suffer from slow convergence rates and poor sample complexity. Hence it is important to design provably sample efficient algorithms for policy optimization. Yet, recent advances for this problems have only been successful in tabular and linear setting, whose benign structures cannot be generalized to non-linearly parameterized policies. In this paper, we address this problem by ","authors_text":"Lin Yang, Yiran Wang, Yu Cheng, Yunfan Li","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-15T23:51:46Z","title":"Low-Switching Policy Gradient with Exploration via Online Sensitivity Sampling"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.09554","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:415a9e188531f93120103ecbbfaa1c641b506f7f44d79cc5c924abb35ec95073","target":"record","created_at":"2026-07-05T06:21:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9eaf3046de77aaec681313b523e261602c7eead8b31ec71e00a30083af1f1307","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-15T23:51:46Z","title_canon_sha256":"4601499b7dec805fcb1b89a97508d7f2bbddc9bf61e3aaef75a51daf03969fe3"},"schema_version":"1.0","source":{"id":"2306.09554","kind":"arxiv","version":1}},"canonical_sha256":"23b83596d7f1dcc4d9aa2582cc72a3e3f491ae862371a86d60c51b17bf6cae63","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"23b83596d7f1dcc4d9aa2582cc72a3e3f491ae862371a86d60c51b17bf6cae63","first_computed_at":"2026-07-05T06:21:21.097300Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:21:21.097300Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"rbMc0pUAjZ9RQ7uSwjCwbHo0DKMkBqCp6zenikux8iGK6CAkl/kKTb+1+73uQItj+2z94IzK2/l0mzitrOPSBg==","signature_status":"signed_v1","signed_at":"2026-07-05T06:21:21.097710Z","signed_message":"canonical_sha256_bytes"},"source_id":"2306.09554","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:415a9e188531f93120103ecbbfaa1c641b506f7f44d79cc5c924abb35ec95073","sha256:16330f7a13429323737e0f60fb771184f8be2af1a0e8b3a442177dab07bfd049"],"state_sha256":"5065a47937393a93c8aa949dffb8453a0b619e58bd5ef46b852a1c71bcf1d7c9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5/iAg2e98U9n/bRaFx5wF0CHgcUcxUXvpbPiHfnvjD0r9J6i3iuIFj3eyznFo5yYa44pID0y3Rf5w+/5Vn2mCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T07:06:14.172455Z","bundle_sha256":"d83e87ea42d8746cae0eef3022ce5afd5d91cea67f844b0ba09805a7fba32815"}}