{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:EM4ZYOEE72N6EKPQ3AQCMKV5SG","short_pith_number":"pith:EM4ZYOEE","canonical_record":{"source":{"id":"2406.12241","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-18T03:32:10Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2564613a1274095dd870a7d70cf2112e88a32dadf52bee7e61a79b87711a51c6","abstract_canon_sha256":"1c2143c2f4c4ccd04c9b9a0c29810c2ac6cf9011807a10e6b9ff0f3e41011010"},"schema_version":"1.0"},"canonical_sha256":"23399c3884fe9be229f0d820262abd91bf76fbfcf44be263bb3ae859c81e7719","source":{"kind":"arxiv","id":"2406.12241","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.12241","created_at":"2026-07-05T08:33:43Z"},{"alias_kind":"arxiv_version","alias_value":"2406.12241v1","created_at":"2026-07-05T08:33:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12241","created_at":"2026-07-05T08:33:43Z"},{"alias_kind":"pith_short_12","alias_value":"EM4ZYOEE72N6","created_at":"2026-07-05T08:33:43Z"},{"alias_kind":"pith_short_16","alias_value":"EM4ZYOEE72N6EKPQ","created_at":"2026-07-05T08:33:43Z"},{"alias_kind":"pith_short_8","alias_value":"EM4ZYOEE","created_at":"2026-07-05T08:33:43Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:EM4ZYOEE72N6EKPQ3AQCMKV5SG","target":"record","payload":{"canonical_record":{"source":{"id":"2406.12241","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-18T03:32:10Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2564613a1274095dd870a7d70cf2112e88a32dadf52bee7e61a79b87711a51c6","abstract_canon_sha256":"1c2143c2f4c4ccd04c9b9a0c29810c2ac6cf9011807a10e6b9ff0f3e41011010"},"schema_version":"1.0"},"canonical_sha256":"23399c3884fe9be229f0d820262abd91bf76fbfcf44be263bb3ae859c81e7719","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:43.192964Z","signature_b64":"pv6taKtyFjlmqCRjCxHd1vn0fxPvZFbaXyDlDBdeSnyl7m5bk2uNiTkJKFVmyz3fdSImYXm+AS3I0vMI9Uu/BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23399c3884fe9be229f0d820262abd91bf76fbfcf44be263bb3ae859c81e7719","last_reissued_at":"2026-07-05T08:33:43.192485Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:43.192485Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2406.12241","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:33:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"3+/BC4CHicLp5MBgpbwcajRbbRzqc0ijvnSTAEy6U77MzpPHzT8JWNdxp2s4u6caZMi/i+tJUbQL9Te4lvE5CQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T10:25:44.407564Z"},"content_sha256":"34f7b82b62be64baa1001d818bcaf0d3a5c2ce8ba9699cc4d4ec3dfd688f06d1","schema_version":"1.0","event_id":"sha256:34f7b82b62be64baa1001d818bcaf0d3a5c2ce8ba9699cc4d4ec3dfd688f06d1"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:EM4ZYOEE72N6EKPQ3AQCMKV5SG","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"More Efficient Randomized Exploration for Reinforcement Learning via Approximate Sampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"A. Rupam Mahmood, Doina Precup, Haque Ishfaq, Jianfeng Lu, Pan Xu, Qingfeng Lan, Yixin Tan, Yu Yang","submitted_at":"2024-06-18T03:32:10Z","abstract_excerpt":"Thompson sampling (TS) is one of the most popular exploration techniques in reinforcement learning (RL). However, most TS algorithms with theoretical guarantees are difficult to implement and not generalizable to Deep RL. While the emerging approximate sampling-based exploration schemes are promising, most existing algorithms are specific to linear Markov Decision Processes (MDP) with suboptimal regret bounds, or only use the most basic samplers such as Langevin Monte Carlo. In this work, we propose an algorithmic framework that incorporates different approximate sampling methods with the rece"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12241","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12241/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:33:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"s+k2ovPpP8z2I/E4JPMZr5n2MkTya2y2HQg1iHJY/Be1V96oHcoEB1fh+oBWKMA7XMG0bF9pebmpNColdsZQDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T10:25:44.407911Z"},"content_sha256":"2ef72b791715227ffbd1481a6d716ce16d267916e8fc44dcd69c3cac24073f6d","schema_version":"1.0","event_id":"sha256:2ef72b791715227ffbd1481a6d716ce16d267916e8fc44dcd69c3cac24073f6d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/EM4ZYOEE72N6EKPQ3AQCMKV5SG/bundle.json","state_url":"https://pith.science/pith/EM4ZYOEE72N6EKPQ3AQCMKV5SG/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/EM4ZYOEE72N6EKPQ3AQCMKV5SG/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T10:25:44Z","links":{"resolver":"https://pith.science/pith/EM4ZYOEE72N6EKPQ3AQCMKV5SG","bundle":"https://pith.science/pith/EM4ZYOEE72N6EKPQ3AQCMKV5SG/bundle.json","state":"https://pith.science/pith/EM4ZYOEE72N6EKPQ3AQCMKV5SG/state.json","well_known_bundle":"https://pith.science/.well-known/pith/EM4ZYOEE72N6EKPQ3AQCMKV5SG/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:EM4ZYOEE72N6EKPQ3AQCMKV5SG","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1c2143c2f4c4ccd04c9b9a0c29810c2ac6cf9011807a10e6b9ff0f3e41011010","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-18T03:32:10Z","title_canon_sha256":"2564613a1274095dd870a7d70cf2112e88a32dadf52bee7e61a79b87711a51c6"},"schema_version":"1.0","source":{"id":"2406.12241","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.12241","created_at":"2026-07-05T08:33:43Z"},{"alias_kind":"arxiv_version","alias_value":"2406.12241v1","created_at":"2026-07-05T08:33:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12241","created_at":"2026-07-05T08:33:43Z"},{"alias_kind":"pith_short_12","alias_value":"EM4ZYOEE72N6","created_at":"2026-07-05T08:33:43Z"},{"alias_kind":"pith_short_16","alias_value":"EM4ZYOEE72N6EKPQ","created_at":"2026-07-05T08:33:43Z"},{"alias_kind":"pith_short_8","alias_value":"EM4ZYOEE","created_at":"2026-07-05T08:33:43Z"}],"graph_snapshots":[{"event_id":"sha256:2ef72b791715227ffbd1481a6d716ce16d267916e8fc44dcd69c3cac24073f6d","target":"graph","created_at":"2026-07-05T08:33:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2406.12241/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Thompson sampling (TS) is one of the most popular exploration techniques in reinforcement learning (RL). However, most TS algorithms with theoretical guarantees are difficult to implement and not generalizable to Deep RL. While the emerging approximate sampling-based exploration schemes are promising, most existing algorithms are specific to linear Markov Decision Processes (MDP) with suboptimal regret bounds, or only use the most basic samplers such as Langevin Monte Carlo. In this work, we propose an algorithmic framework that incorporates different approximate sampling methods with the rece","authors_text":"A. Rupam Mahmood, Doina Precup, Haque Ishfaq, Jianfeng Lu, Pan Xu, Qingfeng Lan, Yixin Tan, Yu Yang","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-18T03:32:10Z","title":"More Efficient Randomized Exploration for Reinforcement Learning via Approximate Sampling"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12241","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:34f7b82b62be64baa1001d818bcaf0d3a5c2ce8ba9699cc4d4ec3dfd688f06d1","target":"record","created_at":"2026-07-05T08:33:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1c2143c2f4c4ccd04c9b9a0c29810c2ac6cf9011807a10e6b9ff0f3e41011010","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-18T03:32:10Z","title_canon_sha256":"2564613a1274095dd870a7d70cf2112e88a32dadf52bee7e61a79b87711a51c6"},"schema_version":"1.0","source":{"id":"2406.12241","kind":"arxiv","version":1}},"canonical_sha256":"23399c3884fe9be229f0d820262abd91bf76fbfcf44be263bb3ae859c81e7719","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"23399c3884fe9be229f0d820262abd91bf76fbfcf44be263bb3ae859c81e7719","first_computed_at":"2026-07-05T08:33:43.192485Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:33:43.192485Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"pv6taKtyFjlmqCRjCxHd1vn0fxPvZFbaXyDlDBdeSnyl7m5bk2uNiTkJKFVmyz3fdSImYXm+AS3I0vMI9Uu/BA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:33:43.192964Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.12241","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:34f7b82b62be64baa1001d818bcaf0d3a5c2ce8ba9699cc4d4ec3dfd688f06d1","sha256:2ef72b791715227ffbd1481a6d716ce16d267916e8fc44dcd69c3cac24073f6d"],"state_sha256":"8886b10a418f8b4f8a50796fbc581ced7e7eab3f2afa00bd796dcbbf1316d7fa"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"mp33oS9C2b+fEMuIibyAymliPmBJYSBb/ArCFpxGz2CwyEcWGc1Co/gs7SfS5yz8u7hsugc9zlsotKXu/lJiDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T10:25:44.411082Z","bundle_sha256":"0d48ecf82e0ee7a50253731cd16cb478dda166730b8012df452d1569342e9bb0"}}