{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:QHN6V3WUMTKRYR3YQ52AIABDHY","short_pith_number":"pith:QHN6V3WU","canonical_record":{"source":{"id":"2404.01598","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-02T02:39:17Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"a3116fd6519ad31fc2b1ebf171040bff961f2546cdee78b63a77782e2d546800","abstract_canon_sha256":"ee7c2bbf76cd0896fd45bde9717d1f915e044c5287a4885fe1d99ea00015a32a"},"schema_version":"1.0"},"canonical_sha256":"81dbeaeed464d51c477887740400233e21f8ad9c03acd4ce181222493218b3f1","source":{"kind":"arxiv","id":"2404.01598","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.01598","created_at":"2026-07-05T08:03:16Z"},{"alias_kind":"arxiv_version","alias_value":"2404.01598v1","created_at":"2026-07-05T08:03:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01598","created_at":"2026-07-05T08:03:16Z"},{"alias_kind":"pith_short_12","alias_value":"QHN6V3WUMTKR","created_at":"2026-07-05T08:03:16Z"},{"alias_kind":"pith_short_16","alias_value":"QHN6V3WUMTKRYR3Y","created_at":"2026-07-05T08:03:16Z"},{"alias_kind":"pith_short_8","alias_value":"QHN6V3WU","created_at":"2026-07-05T08:03:16Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:QHN6V3WUMTKRYR3YQ52AIABDHY","target":"record","payload":{"canonical_record":{"source":{"id":"2404.01598","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-02T02:39:17Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"a3116fd6519ad31fc2b1ebf171040bff961f2546cdee78b63a77782e2d546800","abstract_canon_sha256":"ee7c2bbf76cd0896fd45bde9717d1f915e044c5287a4885fe1d99ea00015a32a"},"schema_version":"1.0"},"canonical_sha256":"81dbeaeed464d51c477887740400233e21f8ad9c03acd4ce181222493218b3f1","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:03:16.476107Z","signature_b64":"rsyrBF8Zoib0Qg4koU38ZbcGoXf9+Gg/ocDFAFPBSl4LF0CyT0YhxZcQQu6MRSc+eQ5WONkbX2gK6860qeR3Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"81dbeaeed464d51c477887740400233e21f8ad9c03acd4ce181222493218b3f1","last_reissued_at":"2026-07-05T08:03:16.475706Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:03:16.475706Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2404.01598","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:03:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0DSLWM2XPKN+p3ALbhuKynhWyRpKUrXsmMzMj9ZOsUnrGRP26XatvsbxvrGAvfTMucGuskWsInN+A8hEc8MPDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T20:26:19.109828Z"},"content_sha256":"5139c4004212adf4136d85d37551e2ad0f093985d5f2d30e5aca92df007feb0c","schema_version":"1.0","event_id":"sha256:5139c4004212adf4136d85d37551e2ad0f093985d5f2d30e5aca92df007feb0c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:QHN6V3WUMTKRYR3YQ52AIABDHY","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Extremum-Seeking Action Selection for Accelerating Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Sicun Gao, Ya-Chien Chang","submitted_at":"2024-04-02T02:39:17Z","abstract_excerpt":"Reinforcement learning for control over continuous spaces typically uses high-entropy stochastic policies, such as Gaussian distributions, for local exploration and estimating policy gradient to optimize performance. Many robotic control problems deal with complex unstable dynamics, where applying actions that are off the feasible control manifolds can quickly lead to undesirable divergence. In such cases, most samples taken from the ambient action space generate low-value trajectories that hardly contribute to policy improvement, resulting in slow or failed learning. We propose to improve act"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01598","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.01598/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:03:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PHQtYTzYPrb2dzNJ6aTW9lZxUN5/5QGOeyJ+HJqH/aKjKfyWlkIsoOKkJLnskN12J27yV8vgVULow/HYsjTPAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T20:26:19.110211Z"},"content_sha256":"7a578b8aae9b0d58ee960e5f127ef56e8bc35fe819f014161a46357cc2d7e23e","schema_version":"1.0","event_id":"sha256:7a578b8aae9b0d58ee960e5f127ef56e8bc35fe819f014161a46357cc2d7e23e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/QHN6V3WUMTKRYR3YQ52AIABDHY/bundle.json","state_url":"https://pith.science/pith/QHN6V3WUMTKRYR3YQ52AIABDHY/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/QHN6V3WUMTKRYR3YQ52AIABDHY/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T20:26:19Z","links":{"resolver":"https://pith.science/pith/QHN6V3WUMTKRYR3YQ52AIABDHY","bundle":"https://pith.science/pith/QHN6V3WUMTKRYR3YQ52AIABDHY/bundle.json","state":"https://pith.science/pith/QHN6V3WUMTKRYR3YQ52AIABDHY/state.json","well_known_bundle":"https://pith.science/.well-known/pith/QHN6V3WUMTKRYR3YQ52AIABDHY/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:QHN6V3WUMTKRYR3YQ52AIABDHY","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ee7c2bbf76cd0896fd45bde9717d1f915e044c5287a4885fe1d99ea00015a32a","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-02T02:39:17Z","title_canon_sha256":"a3116fd6519ad31fc2b1ebf171040bff961f2546cdee78b63a77782e2d546800"},"schema_version":"1.0","source":{"id":"2404.01598","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.01598","created_at":"2026-07-05T08:03:16Z"},{"alias_kind":"arxiv_version","alias_value":"2404.01598v1","created_at":"2026-07-05T08:03:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01598","created_at":"2026-07-05T08:03:16Z"},{"alias_kind":"pith_short_12","alias_value":"QHN6V3WUMTKR","created_at":"2026-07-05T08:03:16Z"},{"alias_kind":"pith_short_16","alias_value":"QHN6V3WUMTKRYR3Y","created_at":"2026-07-05T08:03:16Z"},{"alias_kind":"pith_short_8","alias_value":"QHN6V3WU","created_at":"2026-07-05T08:03:16Z"}],"graph_snapshots":[{"event_id":"sha256:7a578b8aae9b0d58ee960e5f127ef56e8bc35fe819f014161a46357cc2d7e23e","target":"graph","created_at":"2026-07-05T08:03:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2404.01598/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning for control over continuous spaces typically uses high-entropy stochastic policies, such as Gaussian distributions, for local exploration and estimating policy gradient to optimize performance. Many robotic control problems deal with complex unstable dynamics, where applying actions that are off the feasible control manifolds can quickly lead to undesirable divergence. In such cases, most samples taken from the ambient action space generate low-value trajectories that hardly contribute to policy improvement, resulting in slow or failed learning. We propose to improve act","authors_text":"Sicun Gao, Ya-Chien Chang","cross_cats":["cs.AI","cs.RO"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-02T02:39:17Z","title":"Extremum-Seeking Action Selection for Accelerating Policy Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01598","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5139c4004212adf4136d85d37551e2ad0f093985d5f2d30e5aca92df007feb0c","target":"record","created_at":"2026-07-05T08:03:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ee7c2bbf76cd0896fd45bde9717d1f915e044c5287a4885fe1d99ea00015a32a","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-02T02:39:17Z","title_canon_sha256":"a3116fd6519ad31fc2b1ebf171040bff961f2546cdee78b63a77782e2d546800"},"schema_version":"1.0","source":{"id":"2404.01598","kind":"arxiv","version":1}},"canonical_sha256":"81dbeaeed464d51c477887740400233e21f8ad9c03acd4ce181222493218b3f1","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"81dbeaeed464d51c477887740400233e21f8ad9c03acd4ce181222493218b3f1","first_computed_at":"2026-07-05T08:03:16.475706Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:03:16.475706Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"rsyrBF8Zoib0Qg4koU38ZbcGoXf9+Gg/ocDFAFPBSl4LF0CyT0YhxZcQQu6MRSc+eQ5WONkbX2gK6860qeR3Bg==","signature_status":"signed_v1","signed_at":"2026-07-05T08:03:16.476107Z","signed_message":"canonical_sha256_bytes"},"source_id":"2404.01598","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5139c4004212adf4136d85d37551e2ad0f093985d5f2d30e5aca92df007feb0c","sha256:7a578b8aae9b0d58ee960e5f127ef56e8bc35fe819f014161a46357cc2d7e23e"],"state_sha256":"27522951b8bed75a30f45e681f343a76d0ccb36aad2d7c07464857ee285c9d62"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"umHEoqkDlUX6/hJY9OO/2uEflgcJ4MeGKtNTAaP90lVccOXpeYXYjp2F/n7uwL3dhxI7YL7Y7wiEkQuH9WkjCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T20:26:19.113285Z","bundle_sha256":"d6e8eed2c5d470be9a1c324aa9f574948613c208f6007a3b721aff54051dbd16"}}