{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:HILGXV5FVW4GCKPDSLK47SBVXV","short_pith_number":"pith:HILGXV5F","canonical_record":{"source":{"id":"2607.12924","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-14T15:57:25Z","cross_cats_sorted":[],"title_canon_sha256":"674547d35b5df684d2d468890bce7f4e935dff5590e84bd51390c462559752bd","abstract_canon_sha256":"dde5f3015969c132b748eb97fe524845c83f748439572d8cd9753e812ff9d0f7"},"schema_version":"1.0"},"canonical_sha256":"3a166bd7a5adb86129e392d5cfc835bd51b1c94847e4b4ec20addd72f0bdb23f","source":{"kind":"arxiv","id":"2607.12924","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.12924","created_at":"2026-07-15T01:22:25Z"},{"alias_kind":"arxiv_version","alias_value":"2607.12924v1","created_at":"2026-07-15T01:22:25Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.12924","created_at":"2026-07-15T01:22:25Z"},{"alias_kind":"pith_short_12","alias_value":"HILGXV5FVW4G","created_at":"2026-07-15T01:22:25Z"},{"alias_kind":"pith_short_16","alias_value":"HILGXV5FVW4GCKPD","created_at":"2026-07-15T01:22:25Z"},{"alias_kind":"pith_short_8","alias_value":"HILGXV5F","created_at":"2026-07-15T01:22:25Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:HILGXV5FVW4GCKPDSLK47SBVXV","target":"record","payload":{"canonical_record":{"source":{"id":"2607.12924","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-14T15:57:25Z","cross_cats_sorted":[],"title_canon_sha256":"674547d35b5df684d2d468890bce7f4e935dff5590e84bd51390c462559752bd","abstract_canon_sha256":"dde5f3015969c132b748eb97fe524845c83f748439572d8cd9753e812ff9d0f7"},"schema_version":"1.0"},"canonical_sha256":"3a166bd7a5adb86129e392d5cfc835bd51b1c94847e4b4ec20addd72f0bdb23f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-15T01:22:25.420746Z","signature_b64":"PW8C4gl7QHgQ2NOHoA3tTLF8+N6vdiAtduC+zjmR4INUw3caV7tyb6pK6iHrC+khg0k/A6U12AMgKvGYOFJLCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a166bd7a5adb86129e392d5cfc835bd51b1c94847e4b4ec20addd72f0bdb23f","last_reissued_at":"2026-07-15T01:22:25.419640Z","signature_status":"signed_v1","first_computed_at":"2026-07-15T01:22:25.419640Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.12924","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-15T01:22:25Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"DM1DVFBzX95WTeUqyjuytzfcEBFGGzFEBAZxBhJk8ma8ot81Yu2mAVUhO0w6tLUmhTwNHv9NQTPtF3c4wbBtCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T19:40:00.660211Z"},"content_sha256":"ae0b14b08f4c168d71f677611cf6bfe8644ef62dfb9a501e3cf71471482d3f7b","schema_version":"1.0","event_id":"sha256:ae0b14b08f4c168d71f677611cf6bfe8644ef62dfb9a501e3cf71471482d3f7b"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:HILGXV5FVW4GCKPDSLK47SBVXV","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Knowledge- and Gradient-Guided Reinforcement Learning for Parametrized Action Markov Decision Processes","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Jonas Ehrhardt, Oliver Niggemann, Ren\\'e Heesch","submitted_at":"2026-07-14T15:57:25Z","abstract_excerpt":"In this paper, we study Reinforcement Learning in Parametrized Action Markov Decision Processes (PAMDP), where each decision consists of a symbolic action and numerical parameters. In such settings Reinforcement Learning algorithms typically determine parameters with one-shot estimators, which makes their training sample inefficient. Though in most PAMDP environments explicit but incomplete knowledge (e.g., rules, safety constraints, or expert heuristics) is available, it is rarely directly used to increase the sample-efficiency of training Reinforcement Learning agents. We step into this gap "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.12924","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.12924/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-15T01:22:25Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FEwxXKYxgE+3adtWgq4YqtclT7wi/d/Ioai9WTRaSSGIAmOvuOE8ajYBNuQAfrsKcLWGaryctPI3xSDYjOYpDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T19:40:00.660695Z"},"content_sha256":"7e3735dbd0a42d10e859663af6e9f4609a457b7c1cc18d22dcfda88110350ba6","schema_version":"1.0","event_id":"sha256:7e3735dbd0a42d10e859663af6e9f4609a457b7c1cc18d22dcfda88110350ba6"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/HILGXV5FVW4GCKPDSLK47SBVXV/bundle.json","state_url":"https://pith.science/pith/HILGXV5FVW4GCKPDSLK47SBVXV/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/HILGXV5FVW4GCKPDSLK47SBVXV/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-03T19:40:00Z","links":{"resolver":"https://pith.science/pith/HILGXV5FVW4GCKPDSLK47SBVXV","bundle":"https://pith.science/pith/HILGXV5FVW4GCKPDSLK47SBVXV/bundle.json","state":"https://pith.science/pith/HILGXV5FVW4GCKPDSLK47SBVXV/state.json","well_known_bundle":"https://pith.science/.well-known/pith/HILGXV5FVW4GCKPDSLK47SBVXV/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:HILGXV5FVW4GCKPDSLK47SBVXV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"dde5f3015969c132b748eb97fe524845c83f748439572d8cd9753e812ff9d0f7","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-14T15:57:25Z","title_canon_sha256":"674547d35b5df684d2d468890bce7f4e935dff5590e84bd51390c462559752bd"},"schema_version":"1.0","source":{"id":"2607.12924","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.12924","created_at":"2026-07-15T01:22:25Z"},{"alias_kind":"arxiv_version","alias_value":"2607.12924v1","created_at":"2026-07-15T01:22:25Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.12924","created_at":"2026-07-15T01:22:25Z"},{"alias_kind":"pith_short_12","alias_value":"HILGXV5FVW4G","created_at":"2026-07-15T01:22:25Z"},{"alias_kind":"pith_short_16","alias_value":"HILGXV5FVW4GCKPD","created_at":"2026-07-15T01:22:25Z"},{"alias_kind":"pith_short_8","alias_value":"HILGXV5F","created_at":"2026-07-15T01:22:25Z"}],"graph_snapshots":[{"event_id":"sha256:7e3735dbd0a42d10e859663af6e9f4609a457b7c1cc18d22dcfda88110350ba6","target":"graph","created_at":"2026-07-15T01:22:25Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.12924/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In this paper, we study Reinforcement Learning in Parametrized Action Markov Decision Processes (PAMDP), where each decision consists of a symbolic action and numerical parameters. In such settings Reinforcement Learning algorithms typically determine parameters with one-shot estimators, which makes their training sample inefficient. Though in most PAMDP environments explicit but incomplete knowledge (e.g., rules, safety constraints, or expert heuristics) is available, it is rarely directly used to increase the sample-efficiency of training Reinforcement Learning agents. We step into this gap ","authors_text":"Jonas Ehrhardt, Oliver Niggemann, Ren\\'e Heesch","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-14T15:57:25Z","title":"Knowledge- and Gradient-Guided Reinforcement Learning for Parametrized Action Markov Decision Processes"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.12924","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ae0b14b08f4c168d71f677611cf6bfe8644ef62dfb9a501e3cf71471482d3f7b","target":"record","created_at":"2026-07-15T01:22:25Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"dde5f3015969c132b748eb97fe524845c83f748439572d8cd9753e812ff9d0f7","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-14T15:57:25Z","title_canon_sha256":"674547d35b5df684d2d468890bce7f4e935dff5590e84bd51390c462559752bd"},"schema_version":"1.0","source":{"id":"2607.12924","kind":"arxiv","version":1}},"canonical_sha256":"3a166bd7a5adb86129e392d5cfc835bd51b1c94847e4b4ec20addd72f0bdb23f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3a166bd7a5adb86129e392d5cfc835bd51b1c94847e4b4ec20addd72f0bdb23f","first_computed_at":"2026-07-15T01:22:25.419640Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-15T01:22:25.419640Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"PW8C4gl7QHgQ2NOHoA3tTLF8+N6vdiAtduC+zjmR4INUw3caV7tyb6pK6iHrC+khg0k/A6U12AMgKvGYOFJLCg==","signature_status":"signed_v1","signed_at":"2026-07-15T01:22:25.420746Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.12924","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ae0b14b08f4c168d71f677611cf6bfe8644ef62dfb9a501e3cf71471482d3f7b","sha256:7e3735dbd0a42d10e859663af6e9f4609a457b7c1cc18d22dcfda88110350ba6"],"state_sha256":"41d02438b81cfe7ad0d485bf917fa627627e627bda646148b0aeae88b532e339"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Wbn/w1eIrBcgKVlCn5Xi9xLYwgzZzNYjPZl8+1R230xxZeZZ2Rgss47LhXF7SKpMoPWn3m8E1AmhOt15Pk4NBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-03T19:40:00.665446Z","bundle_sha256":"802b5caf9faf056803f3ba514ebbe5807be8409c1b109ec6737b5434997f6331"}}