{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:RDHGEB4UIVHDHMN7FITQGFQQOC","short_pith_number":"pith:RDHGEB4U","canonical_record":{"source":{"id":"2407.09905","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-13T14:45:08Z","cross_cats_sorted":[],"title_canon_sha256":"67b1f6da1c70a30e10c55b39804052043fe098cb3d46ddbf74366d6f16be637a","abstract_canon_sha256":"c1ba7cd077151a2db58910b669263a1e8c05bf5eb09aa9dcb08021d71b243e78"},"schema_version":"1.0"},"canonical_sha256":"88ce620794454e33b1bf2a2703161070a0e4ddea9142092a75e158e0f1bef160","source":{"kind":"arxiv","id":"2407.09905","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2407.09905","created_at":"2026-07-05T08:43:36Z"},{"alias_kind":"arxiv_version","alias_value":"2407.09905v1","created_at":"2026-07-05T08:43:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09905","created_at":"2026-07-05T08:43:36Z"},{"alias_kind":"pith_short_12","alias_value":"RDHGEB4UIVHD","created_at":"2026-07-05T08:43:36Z"},{"alias_kind":"pith_short_16","alias_value":"RDHGEB4UIVHDHMN7","created_at":"2026-07-05T08:43:36Z"},{"alias_kind":"pith_short_8","alias_value":"RDHGEB4U","created_at":"2026-07-05T08:43:36Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:RDHGEB4UIVHDHMN7FITQGFQQOC","target":"record","payload":{"canonical_record":{"source":{"id":"2407.09905","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-13T14:45:08Z","cross_cats_sorted":[],"title_canon_sha256":"67b1f6da1c70a30e10c55b39804052043fe098cb3d46ddbf74366d6f16be637a","abstract_canon_sha256":"c1ba7cd077151a2db58910b669263a1e8c05bf5eb09aa9dcb08021d71b243e78"},"schema_version":"1.0"},"canonical_sha256":"88ce620794454e33b1bf2a2703161070a0e4ddea9142092a75e158e0f1bef160","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:36.400267Z","signature_b64":"ZE3CxWU1PvzB1GbtU29MuzaEzJ7h0MqyEG6aqZzkAA9TfUc21JhaTo9ZeRg+hS4cNcTayuuLQracjTlJ74ntCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"88ce620794454e33b1bf2a2703161070a0e4ddea9142092a75e158e0f1bef160","last_reissued_at":"2026-07-05T08:43:36.399908Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:36.399908Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2407.09905","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:43:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Bfi+YnLzmuH2w2yr7TbLGmxlKgID50KtIyT+7wUGbK/HsIIlxLynxgcbMTAth+b+u3xNEfwGHx/toos8/7adAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T15:42:33.579340Z"},"content_sha256":"5261ffba9eaecd5a4564f5ab41861084b42e5acbdce9f7cda93364faacf80098","schema_version":"1.0","event_id":"sha256:5261ffba9eaecd5a4564f5ab41861084b42e5acbdce9f7cda93364faacf80098"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:RDHGEB4UIVHDHMN7FITQGFQQOC","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Global Reinforcement Learning: Beyond Linear and Convex Rewards via Submodular Semi-gradient Methods","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andreas Krause, Manish Prajapat, Riccardo De Santi","submitted_at":"2024-07-13T14:45:08Z","abstract_excerpt":"In classic Reinforcement Learning (RL), the agent maximizes an additive objective of the visited states, e.g., a value function. Unfortunately, objectives of this type cannot model many real-world applications such as experiment design, exploration, imitation learning, and risk-averse RL to name a few. This is due to the fact that additive objectives disregard interactions between states that are crucial for certain tasks. To tackle this problem, we introduce Global RL (GRL), where rewards are globally defined over trajectories instead of locally over states. Global rewards can capture negativ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09905","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.09905/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:43:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"yuo3lHtDh7p7PZo+Op2L7hAEd+1aSzuDPvgWLQTew4SD0UTrwR2yzW+mszFrPs+Pn5SxWq9CmsdBJVkbbBD2AA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T15:42:33.580059Z"},"content_sha256":"fd2363aa867f1fb81af3bf4ff762eb0d4e264c4ad330dabcec67cc971bc53a41","schema_version":"1.0","event_id":"sha256:fd2363aa867f1fb81af3bf4ff762eb0d4e264c4ad330dabcec67cc971bc53a41"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RDHGEB4UIVHDHMN7FITQGFQQOC/bundle.json","state_url":"https://pith.science/pith/RDHGEB4UIVHDHMN7FITQGFQQOC/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RDHGEB4UIVHDHMN7FITQGFQQOC/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-18T15:42:33Z","links":{"resolver":"https://pith.science/pith/RDHGEB4UIVHDHMN7FITQGFQQOC","bundle":"https://pith.science/pith/RDHGEB4UIVHDHMN7FITQGFQQOC/bundle.json","state":"https://pith.science/pith/RDHGEB4UIVHDHMN7FITQGFQQOC/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RDHGEB4UIVHDHMN7FITQGFQQOC/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:RDHGEB4UIVHDHMN7FITQGFQQOC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c1ba7cd077151a2db58910b669263a1e8c05bf5eb09aa9dcb08021d71b243e78","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-13T14:45:08Z","title_canon_sha256":"67b1f6da1c70a30e10c55b39804052043fe098cb3d46ddbf74366d6f16be637a"},"schema_version":"1.0","source":{"id":"2407.09905","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2407.09905","created_at":"2026-07-05T08:43:36Z"},{"alias_kind":"arxiv_version","alias_value":"2407.09905v1","created_at":"2026-07-05T08:43:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09905","created_at":"2026-07-05T08:43:36Z"},{"alias_kind":"pith_short_12","alias_value":"RDHGEB4UIVHD","created_at":"2026-07-05T08:43:36Z"},{"alias_kind":"pith_short_16","alias_value":"RDHGEB4UIVHDHMN7","created_at":"2026-07-05T08:43:36Z"},{"alias_kind":"pith_short_8","alias_value":"RDHGEB4U","created_at":"2026-07-05T08:43:36Z"}],"graph_snapshots":[{"event_id":"sha256:fd2363aa867f1fb81af3bf4ff762eb0d4e264c4ad330dabcec67cc971bc53a41","target":"graph","created_at":"2026-07-05T08:43:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2407.09905/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In classic Reinforcement Learning (RL), the agent maximizes an additive objective of the visited states, e.g., a value function. Unfortunately, objectives of this type cannot model many real-world applications such as experiment design, exploration, imitation learning, and risk-averse RL to name a few. This is due to the fact that additive objectives disregard interactions between states that are crucial for certain tasks. To tackle this problem, we introduce Global RL (GRL), where rewards are globally defined over trajectories instead of locally over states. Global rewards can capture negativ","authors_text":"Andreas Krause, Manish Prajapat, Riccardo De Santi","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-13T14:45:08Z","title":"Global Reinforcement Learning: Beyond Linear and Convex Rewards via Submodular Semi-gradient Methods"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09905","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5261ffba9eaecd5a4564f5ab41861084b42e5acbdce9f7cda93364faacf80098","target":"record","created_at":"2026-07-05T08:43:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c1ba7cd077151a2db58910b669263a1e8c05bf5eb09aa9dcb08021d71b243e78","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-13T14:45:08Z","title_canon_sha256":"67b1f6da1c70a30e10c55b39804052043fe098cb3d46ddbf74366d6f16be637a"},"schema_version":"1.0","source":{"id":"2407.09905","kind":"arxiv","version":1}},"canonical_sha256":"88ce620794454e33b1bf2a2703161070a0e4ddea9142092a75e158e0f1bef160","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"88ce620794454e33b1bf2a2703161070a0e4ddea9142092a75e158e0f1bef160","first_computed_at":"2026-07-05T08:43:36.399908Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:43:36.399908Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ZE3CxWU1PvzB1GbtU29MuzaEzJ7h0MqyEG6aqZzkAA9TfUc21JhaTo9ZeRg+hS4cNcTayuuLQracjTlJ74ntCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T08:43:36.400267Z","signed_message":"canonical_sha256_bytes"},"source_id":"2407.09905","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5261ffba9eaecd5a4564f5ab41861084b42e5acbdce9f7cda93364faacf80098","sha256:fd2363aa867f1fb81af3bf4ff762eb0d4e264c4ad330dabcec67cc971bc53a41"],"state_sha256":"e2f07096b6085d735092951ab0fb9d3aec7aac5b1bbbdb7a6410bfb680a21a80"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"hndbHfy2TPHgnD4UnTfBkRGv5PyhR/tdT9GvIUfjpy1oNHaThqEOUCO2tpZFwwFyF7l7Fyb8ENlFgZAI/40ACw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-18T15:42:33.589437Z","bundle_sha256":"69c3f8fc9351aa0eefe46ce681c00335a895e15d12abd7a831619210fe77eaa0"}}