{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:H4RJVQ4UCT3TLXOHU3AFYIYY3S","short_pith_number":"pith:H4RJVQ4U","canonical_record":{"source":{"id":"2607.27968","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T10:15:59Z","cross_cats_sorted":[],"title_canon_sha256":"04c364beac1f4c1e713b7fd419cbd6228ada5dc9768a58df400ddc33b00758ac","abstract_canon_sha256":"f307a25943068186b904ba18255906bc6339c5ce2eb42db72115dc2d2ed1f9ea"},"schema_version":"1.0"},"canonical_sha256":"3f229ac39414f735ddc7a6c05c2318dc82a4c4e39312f87c2ccd917aede77ded","source":{"kind":"arxiv","id":"2607.27968","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.27968","created_at":"2026-07-31T01:35:08Z"},{"alias_kind":"arxiv_version","alias_value":"2607.27968v1","created_at":"2026-07-31T01:35:08Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.27968","created_at":"2026-07-31T01:35:08Z"},{"alias_kind":"pith_short_12","alias_value":"H4RJVQ4UCT3T","created_at":"2026-07-31T01:35:08Z"},{"alias_kind":"pith_short_16","alias_value":"H4RJVQ4UCT3TLXOH","created_at":"2026-07-31T01:35:08Z"},{"alias_kind":"pith_short_8","alias_value":"H4RJVQ4U","created_at":"2026-07-31T01:35:08Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:H4RJVQ4UCT3TLXOHU3AFYIYY3S","target":"record","payload":{"canonical_record":{"source":{"id":"2607.27968","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T10:15:59Z","cross_cats_sorted":[],"title_canon_sha256":"04c364beac1f4c1e713b7fd419cbd6228ada5dc9768a58df400ddc33b00758ac","abstract_canon_sha256":"f307a25943068186b904ba18255906bc6339c5ce2eb42db72115dc2d2ed1f9ea"},"schema_version":"1.0"},"canonical_sha256":"3f229ac39414f735ddc7a6c05c2318dc82a4c4e39312f87c2ccd917aede77ded","receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f229ac39414f735ddc7a6c05c2318dc82a4c4e39312f87c2ccd917aede77ded","last_reissued_at":"2026-07-31T01:35:08.803957Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-31T01:35:08.803957Z"},"source_kind":"arxiv","source_id":"2607.27968","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-31T01:35:08Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"DrO8dwc3fZOEClL7SorE2M44t5lxpeuCe7RpYOxtq9JTVy4ta9bqzLQkvt1KsuIPmWF1DxuSgyBU4itmzc0DCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T19:53:27.391502Z"},"content_sha256":"19ba5b79f5819f5283daaf098b569be4d7a59c6854c42e53223499ebdfbc68f4","schema_version":"1.0","event_id":"sha256:19ba5b79f5819f5283daaf098b569be4d7a59c6854c42e53223499ebdfbc68f4"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:H4RJVQ4UCT3TLXOHU3AFYIYY3S","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Beyond Binary Rewards: A Comparative Study of Reward Design for Reinforcement Unlearning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bardh Prenkaj, Davide Gabrielli, Efstratios Zaradoukas, Gjergji Kasneci","submitted_at":"2026-07-30T10:15:59Z","abstract_excerpt":"Machine unlearning seeks to selectively remove specific knowledge from trained language models without full retraining, a growing necessity under privacy regulations such as GDPR and the EU AI Act. Recent work has reformulated unlearning as a Reinforcement Learning with Verifiable Rewards (RLVR) problem, where models are optimized against verifiable rewards computed directly from their outputs. However, existing methods rely on sparse binary rewards that provide minimal learning signal, indicating only whether forbidden content was avoided, and limiting convergence speed. In this paper, we stu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.27968","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.27968/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-31T01:35:08Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zeAaBZLntuT7yu/kA7OfGp8fdyuruh5na87lfcq8TTZ/Dig7gZKKq9ayCaq19fqoOunczJMa+C0xqmSTPLTPBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T19:53:27.392132Z"},"content_sha256":"25bcde3647ac4350ca7b0d48d92edd57c57c01a6296b1576dc9e9783df4566d4","schema_version":"1.0","event_id":"sha256:25bcde3647ac4350ca7b0d48d92edd57c57c01a6296b1576dc9e9783df4566d4"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/H4RJVQ4UCT3TLXOHU3AFYIYY3S/bundle.json","state_url":"https://pith.science/pith/H4RJVQ4UCT3TLXOHU3AFYIYY3S/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/H4RJVQ4UCT3TLXOHU3AFYIYY3S/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T19:53:27Z","links":{"resolver":"https://pith.science/pith/H4RJVQ4UCT3TLXOHU3AFYIYY3S","bundle":"https://pith.science/pith/H4RJVQ4UCT3TLXOHU3AFYIYY3S/bundle.json","state":"https://pith.science/pith/H4RJVQ4UCT3TLXOHU3AFYIYY3S/state.json","well_known_bundle":"https://pith.science/.well-known/pith/H4RJVQ4UCT3TLXOHU3AFYIYY3S/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:H4RJVQ4UCT3TLXOHU3AFYIYY3S","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f307a25943068186b904ba18255906bc6339c5ce2eb42db72115dc2d2ed1f9ea","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T10:15:59Z","title_canon_sha256":"04c364beac1f4c1e713b7fd419cbd6228ada5dc9768a58df400ddc33b00758ac"},"schema_version":"1.0","source":{"id":"2607.27968","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.27968","created_at":"2026-07-31T01:35:08Z"},{"alias_kind":"arxiv_version","alias_value":"2607.27968v1","created_at":"2026-07-31T01:35:08Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.27968","created_at":"2026-07-31T01:35:08Z"},{"alias_kind":"pith_short_12","alias_value":"H4RJVQ4UCT3T","created_at":"2026-07-31T01:35:08Z"},{"alias_kind":"pith_short_16","alias_value":"H4RJVQ4UCT3TLXOH","created_at":"2026-07-31T01:35:08Z"},{"alias_kind":"pith_short_8","alias_value":"H4RJVQ4U","created_at":"2026-07-31T01:35:08Z"}],"graph_snapshots":[{"event_id":"sha256:25bcde3647ac4350ca7b0d48d92edd57c57c01a6296b1576dc9e9783df4566d4","target":"graph","created_at":"2026-07-31T01:35:08Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.27968/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Machine unlearning seeks to selectively remove specific knowledge from trained language models without full retraining, a growing necessity under privacy regulations such as GDPR and the EU AI Act. Recent work has reformulated unlearning as a Reinforcement Learning with Verifiable Rewards (RLVR) problem, where models are optimized against verifiable rewards computed directly from their outputs. However, existing methods rely on sparse binary rewards that provide minimal learning signal, indicating only whether forbidden content was avoided, and limiting convergence speed. In this paper, we stu","authors_text":"Bardh Prenkaj, Davide Gabrielli, Efstratios Zaradoukas, Gjergji Kasneci","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T10:15:59Z","title":"Beyond Binary Rewards: A Comparative Study of Reward Design for Reinforcement Unlearning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.27968","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:19ba5b79f5819f5283daaf098b569be4d7a59c6854c42e53223499ebdfbc68f4","target":"record","created_at":"2026-07-31T01:35:08Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f307a25943068186b904ba18255906bc6339c5ce2eb42db72115dc2d2ed1f9ea","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T10:15:59Z","title_canon_sha256":"04c364beac1f4c1e713b7fd419cbd6228ada5dc9768a58df400ddc33b00758ac"},"schema_version":"1.0","source":{"id":"2607.27968","kind":"arxiv","version":1}},"canonical_sha256":"3f229ac39414f735ddc7a6c05c2318dc82a4c4e39312f87c2ccd917aede77ded","receipt":{"builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3f229ac39414f735ddc7a6c05c2318dc82a4c4e39312f87c2ccd917aede77ded","first_computed_at":"2026-07-31T01:35:08.803957Z","kind":"pith_receipt","last_reissued_at":"2026-07-31T01:35:08.803957Z","receipt_version":"0.3","signature_status":"unsigned_v0"},"source_id":"2607.27968","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:19ba5b79f5819f5283daaf098b569be4d7a59c6854c42e53223499ebdfbc68f4","sha256:25bcde3647ac4350ca7b0d48d92edd57c57c01a6296b1576dc9e9783df4566d4"],"state_sha256":"b08a0f6d9ab3e1772cbbf1b7e33e411ef90e962d11efbe289285f8b73297e7ce"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+RVeNS6wO7aFXeLqkuOahjoZ250GkW5nMURx4OHYW4ZFjC/ItuiR2eFIpVeN53kmABbJByzPvqTFEr+bC1XLCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T19:53:27.398284Z","bundle_sha256":"31f3d38416a8c8e1810654618213a23b5fc09b8610f45d5c9db38d42c8f3443a"}}