{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:UCNRFLIU26X7VZNTL2N7MFLKXU","short_pith_number":"pith:UCNRFLIU","canonical_record":{"source":{"id":"2502.16328","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-22T19:14:36Z","cross_cats_sorted":[],"title_canon_sha256":"ec9d13a86f068649f15efc432106f6b995e942b2f92b155191c3fb62f6237c99","abstract_canon_sha256":"3890c705035f3ed2d8bb9b91d29fcb8690e3f9b246bbca2e8a8d5f5936b4d3a5"},"schema_version":"1.0"},"canonical_sha256":"a09b12ad14d7affae5b35e9bf6156abd2b363a704c765eb8d1499cba82adcf36","source":{"kind":"arxiv","id":"2502.16328","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.16328","created_at":"2026-07-05T11:21:43Z"},{"alias_kind":"arxiv_version","alias_value":"2502.16328v2","created_at":"2026-07-05T11:21:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16328","created_at":"2026-07-05T11:21:43Z"},{"alias_kind":"pith_short_12","alias_value":"UCNRFLIU26X7","created_at":"2026-07-05T11:21:43Z"},{"alias_kind":"pith_short_16","alias_value":"UCNRFLIU26X7VZNT","created_at":"2026-07-05T11:21:43Z"},{"alias_kind":"pith_short_8","alias_value":"UCNRFLIU","created_at":"2026-07-05T11:21:43Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:UCNRFLIU26X7VZNTL2N7MFLKXU","target":"record","payload":{"canonical_record":{"source":{"id":"2502.16328","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-22T19:14:36Z","cross_cats_sorted":[],"title_canon_sha256":"ec9d13a86f068649f15efc432106f6b995e942b2f92b155191c3fb62f6237c99","abstract_canon_sha256":"3890c705035f3ed2d8bb9b91d29fcb8690e3f9b246bbca2e8a8d5f5936b4d3a5"},"schema_version":"1.0"},"canonical_sha256":"a09b12ad14d7affae5b35e9bf6156abd2b363a704c765eb8d1499cba82adcf36","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:43.006344Z","signature_b64":"Hwg8J3KRW5MdZ2nOnFq9APNZaT1jPxQwwcCTFD5ygtKrOlktMbBA57fYluRsr4xzV8cSl8466cLNcY1+P6IiDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a09b12ad14d7affae5b35e9bf6156abd2b363a704c765eb8d1499cba82adcf36","last_reissued_at":"2026-07-05T11:21:43.005860Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:43.005860Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.16328","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:21:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Ybl1bkYI6xiaeb8Y8Bjj07xKbYBNOVG9hp2/ZBJ1t/cNrIBAIRmmSdGp05Hz0WMU33OnDTAQRrk1ITd0DqvQDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T14:38:06.664316Z"},"content_sha256":"c00bb0dc94fcdec97fa282410cd4df92c84a017a047b6f9ff40627d29dba2d1a","schema_version":"1.0","event_id":"sha256:c00bb0dc94fcdec97fa282410cd4df92c84a017a047b6f9ff40627d29dba2d1a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:UCNRFLIU26X7VZNTL2N7MFLKXU","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Optimal Transport-Guided Safety in Temporal Difference Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ali Baheri, Zahra Shahrooei","submitted_at":"2025-02-22T19:14:36Z","abstract_excerpt":"The primary goal of reinforcement learning is to develop decision-making policies that prioritize optimal performance, frequently without considering safety. In contrast, safe reinforcement learning seeks to reduce or avoid unsafe behavior. This paper views safety as taking actions with more predictable consequences under environment stochasticity and introduces a temporal difference algorithm that uses optimal transport theory to quantify the uncertainty associated with actions. By integrating this uncertainty score into the decision-making objective, the agent is encouraged to favor actions "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16328","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.16328/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:21:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/C1t5efD3xJeBmrUELTE2QuNgstp2z2PraV9xJebGqcZC7NM9oYulD4rcCnY7XDCq8jbXbbZK5obY97xiJ32CA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T14:38:06.664886Z"},"content_sha256":"f3975688e38bc3e159c6aa95de927a6b5e83f148cba6b3116b0d338bb307222a","schema_version":"1.0","event_id":"sha256:f3975688e38bc3e159c6aa95de927a6b5e83f148cba6b3116b0d338bb307222a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/UCNRFLIU26X7VZNTL2N7MFLKXU/bundle.json","state_url":"https://pith.science/pith/UCNRFLIU26X7VZNTL2N7MFLKXU/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/UCNRFLIU26X7VZNTL2N7MFLKXU/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T14:38:06Z","links":{"resolver":"https://pith.science/pith/UCNRFLIU26X7VZNTL2N7MFLKXU","bundle":"https://pith.science/pith/UCNRFLIU26X7VZNTL2N7MFLKXU/bundle.json","state":"https://pith.science/pith/UCNRFLIU26X7VZNTL2N7MFLKXU/state.json","well_known_bundle":"https://pith.science/.well-known/pith/UCNRFLIU26X7VZNTL2N7MFLKXU/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:UCNRFLIU26X7VZNTL2N7MFLKXU","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"3890c705035f3ed2d8bb9b91d29fcb8690e3f9b246bbca2e8a8d5f5936b4d3a5","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-22T19:14:36Z","title_canon_sha256":"ec9d13a86f068649f15efc432106f6b995e942b2f92b155191c3fb62f6237c99"},"schema_version":"1.0","source":{"id":"2502.16328","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.16328","created_at":"2026-07-05T11:21:43Z"},{"alias_kind":"arxiv_version","alias_value":"2502.16328v2","created_at":"2026-07-05T11:21:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16328","created_at":"2026-07-05T11:21:43Z"},{"alias_kind":"pith_short_12","alias_value":"UCNRFLIU26X7","created_at":"2026-07-05T11:21:43Z"},{"alias_kind":"pith_short_16","alias_value":"UCNRFLIU26X7VZNT","created_at":"2026-07-05T11:21:43Z"},{"alias_kind":"pith_short_8","alias_value":"UCNRFLIU","created_at":"2026-07-05T11:21:43Z"}],"graph_snapshots":[{"event_id":"sha256:f3975688e38bc3e159c6aa95de927a6b5e83f148cba6b3116b0d338bb307222a","target":"graph","created_at":"2026-07-05T11:21:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.16328/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The primary goal of reinforcement learning is to develop decision-making policies that prioritize optimal performance, frequently without considering safety. In contrast, safe reinforcement learning seeks to reduce or avoid unsafe behavior. This paper views safety as taking actions with more predictable consequences under environment stochasticity and introduces a temporal difference algorithm that uses optimal transport theory to quantify the uncertainty associated with actions. By integrating this uncertainty score into the decision-making objective, the agent is encouraged to favor actions ","authors_text":"Ali Baheri, Zahra Shahrooei","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-22T19:14:36Z","title":"Optimal Transport-Guided Safety in Temporal Difference Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16328","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c00bb0dc94fcdec97fa282410cd4df92c84a017a047b6f9ff40627d29dba2d1a","target":"record","created_at":"2026-07-05T11:21:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"3890c705035f3ed2d8bb9b91d29fcb8690e3f9b246bbca2e8a8d5f5936b4d3a5","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-22T19:14:36Z","title_canon_sha256":"ec9d13a86f068649f15efc432106f6b995e942b2f92b155191c3fb62f6237c99"},"schema_version":"1.0","source":{"id":"2502.16328","kind":"arxiv","version":2}},"canonical_sha256":"a09b12ad14d7affae5b35e9bf6156abd2b363a704c765eb8d1499cba82adcf36","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"a09b12ad14d7affae5b35e9bf6156abd2b363a704c765eb8d1499cba82adcf36","first_computed_at":"2026-07-05T11:21:43.005860Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:21:43.005860Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Hwg8J3KRW5MdZ2nOnFq9APNZaT1jPxQwwcCTFD5ygtKrOlktMbBA57fYluRsr4xzV8cSl8466cLNcY1+P6IiDA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:21:43.006344Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.16328","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c00bb0dc94fcdec97fa282410cd4df92c84a017a047b6f9ff40627d29dba2d1a","sha256:f3975688e38bc3e159c6aa95de927a6b5e83f148cba6b3116b0d338bb307222a"],"state_sha256":"05a9422632b0ead75a197fba4ad92619775208c76e1ead167bbdc9b72adf328b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4A5hdf7oFEBVEdYTg6D4jN79qfzMnPwtaibCTdh0ph5tUixE+DBBmYhJemawbidAZUO8OjBip5Shuy1WDxBQBQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T14:38:06.672225Z","bundle_sha256":"0b17018da330799331a60c5754f0c62f9518f77ba0f73ece6dfaa98a82385aef"}}