{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:ZBO6VUH5UYIHUSOKX6LTYXASVG","short_pith_number":"pith:ZBO6VUH5","canonical_record":{"source":{"id":"2301.07421","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-01-18T10:42:00Z","cross_cats_sorted":[],"title_canon_sha256":"8ec22c13d5f627c7dec77c4a8a90b5bf3305a7bfde0875876acd71c0567c9688","abstract_canon_sha256":"2db76f23db0baa2c7cfbc078b265577e9f4138a7fee00fdc77931b1964da2a11"},"schema_version":"1.0"},"canonical_sha256":"c85dead0fda6107a49cabf973c5c12a9b0f0a84366787f0c6cb67ceb729c8809","source":{"kind":"arxiv","id":"2301.07421","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2301.07421","created_at":"2026-07-05T05:34:08Z"},{"alias_kind":"arxiv_version","alias_value":"2301.07421v1","created_at":"2026-07-05T05:34:08Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.07421","created_at":"2026-07-05T05:34:08Z"},{"alias_kind":"pith_short_12","alias_value":"ZBO6VUH5UYIH","created_at":"2026-07-05T05:34:08Z"},{"alias_kind":"pith_short_16","alias_value":"ZBO6VUH5UYIHUSOK","created_at":"2026-07-05T05:34:08Z"},{"alias_kind":"pith_short_8","alias_value":"ZBO6VUH5","created_at":"2026-07-05T05:34:08Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:ZBO6VUH5UYIHUSOKX6LTYXASVG","target":"record","payload":{"canonical_record":{"source":{"id":"2301.07421","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-01-18T10:42:00Z","cross_cats_sorted":[],"title_canon_sha256":"8ec22c13d5f627c7dec77c4a8a90b5bf3305a7bfde0875876acd71c0567c9688","abstract_canon_sha256":"2db76f23db0baa2c7cfbc078b265577e9f4138a7fee00fdc77931b1964da2a11"},"schema_version":"1.0"},"canonical_sha256":"c85dead0fda6107a49cabf973c5c12a9b0f0a84366787f0c6cb67ceb729c8809","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:34:08.951174Z","signature_b64":"7mEq0N/CADuEYK3ZzWXwm2RlEZAJSbqnxfX6ykyrNhvJxJnXAIocgvgtyNg8dJcHaF7HBm3IQxahwAXPl62LDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c85dead0fda6107a49cabf973c5c12a9b0f0a84366787f0c6cb67ceb729c8809","last_reissued_at":"2026-07-05T05:34:08.950752Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:34:08.950752Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2301.07421","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:34:08Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"G9Fpk6yHI1ddq4Y297zlWTX8JYHUkO5vi/8a8pellJIFfnOUf+4D83/4wwZC9dEU679C8MIEEFcKcBP0z+S7DQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T00:32:05.998367Z"},"content_sha256":"6414d996aa4bc1fc5c7341f354ee5b7592bf151b98dd8e6f0a08f8e99e20e260","schema_version":"1.0","event_id":"sha256:6414d996aa4bc1fc5c7341f354ee5b7592bf151b98dd8e6f0a08f8e99e20e260"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:ZBO6VUH5UYIHUSOKX6LTYXASVG","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"DIRECT: Learning from Sparse and Shifting Rewards using Discriminative Reward Co-Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Claudia Linnhoff-Popien, Fabian Ritz, Philipp Altmann, Thomas Gabor, Thomy Phan","submitted_at":"2023-01-18T10:42:00Z","abstract_excerpt":"We propose discriminative reward co-training (DIRECT) as an extension to deep reinforcement learning algorithms. Building upon the concept of self-imitation learning (SIL), we introduce an imitation buffer to store beneficial trajectories generated by the policy determined by their return. A discriminator network is trained concurrently to the policy to distinguish between trajectories generated by the current policy and beneficial trajectories generated by previous policies. The discriminator's verdict is used to construct a reward signal for optimizing the policy. By interpolating prior expe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.07421","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.07421/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:34:08Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"TqNVAtHuutUjp32aX507xK/GifzQAjqg02JcFA0XtdTpsOd3lYXCDNVJcRdC32IPqoP+ju5kVp435XvJrCboAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T00:32:05.999034Z"},"content_sha256":"8bc270029413a39c49a2e1007984f1151142732a5da1f27d44375d2164680770","schema_version":"1.0","event_id":"sha256:8bc270029413a39c49a2e1007984f1151142732a5da1f27d44375d2164680770"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/ZBO6VUH5UYIHUSOKX6LTYXASVG/bundle.json","state_url":"https://pith.science/pith/ZBO6VUH5UYIHUSOKX6LTYXASVG/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/ZBO6VUH5UYIHUSOKX6LTYXASVG/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-01T00:32:06Z","links":{"resolver":"https://pith.science/pith/ZBO6VUH5UYIHUSOKX6LTYXASVG","bundle":"https://pith.science/pith/ZBO6VUH5UYIHUSOKX6LTYXASVG/bundle.json","state":"https://pith.science/pith/ZBO6VUH5UYIHUSOKX6LTYXASVG/state.json","well_known_bundle":"https://pith.science/.well-known/pith/ZBO6VUH5UYIHUSOKX6LTYXASVG/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:ZBO6VUH5UYIHUSOKX6LTYXASVG","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2db76f23db0baa2c7cfbc078b265577e9f4138a7fee00fdc77931b1964da2a11","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-01-18T10:42:00Z","title_canon_sha256":"8ec22c13d5f627c7dec77c4a8a90b5bf3305a7bfde0875876acd71c0567c9688"},"schema_version":"1.0","source":{"id":"2301.07421","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2301.07421","created_at":"2026-07-05T05:34:08Z"},{"alias_kind":"arxiv_version","alias_value":"2301.07421v1","created_at":"2026-07-05T05:34:08Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.07421","created_at":"2026-07-05T05:34:08Z"},{"alias_kind":"pith_short_12","alias_value":"ZBO6VUH5UYIH","created_at":"2026-07-05T05:34:08Z"},{"alias_kind":"pith_short_16","alias_value":"ZBO6VUH5UYIHUSOK","created_at":"2026-07-05T05:34:08Z"},{"alias_kind":"pith_short_8","alias_value":"ZBO6VUH5","created_at":"2026-07-05T05:34:08Z"}],"graph_snapshots":[{"event_id":"sha256:8bc270029413a39c49a2e1007984f1151142732a5da1f27d44375d2164680770","target":"graph","created_at":"2026-07-05T05:34:08Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2301.07421/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We propose discriminative reward co-training (DIRECT) as an extension to deep reinforcement learning algorithms. Building upon the concept of self-imitation learning (SIL), we introduce an imitation buffer to store beneficial trajectories generated by the policy determined by their return. A discriminator network is trained concurrently to the policy to distinguish between trajectories generated by the current policy and beneficial trajectories generated by previous policies. The discriminator's verdict is used to construct a reward signal for optimizing the policy. By interpolating prior expe","authors_text":"Claudia Linnhoff-Popien, Fabian Ritz, Philipp Altmann, Thomas Gabor, Thomy Phan","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-01-18T10:42:00Z","title":"DIRECT: Learning from Sparse and Shifting Rewards using Discriminative Reward Co-Training"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.07421","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:6414d996aa4bc1fc5c7341f354ee5b7592bf151b98dd8e6f0a08f8e99e20e260","target":"record","created_at":"2026-07-05T05:34:08Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2db76f23db0baa2c7cfbc078b265577e9f4138a7fee00fdc77931b1964da2a11","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-01-18T10:42:00Z","title_canon_sha256":"8ec22c13d5f627c7dec77c4a8a90b5bf3305a7bfde0875876acd71c0567c9688"},"schema_version":"1.0","source":{"id":"2301.07421","kind":"arxiv","version":1}},"canonical_sha256":"c85dead0fda6107a49cabf973c5c12a9b0f0a84366787f0c6cb67ceb729c8809","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"c85dead0fda6107a49cabf973c5c12a9b0f0a84366787f0c6cb67ceb729c8809","first_computed_at":"2026-07-05T05:34:08.950752Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:34:08.950752Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"7mEq0N/CADuEYK3ZzWXwm2RlEZAJSbqnxfX6ykyrNhvJxJnXAIocgvgtyNg8dJcHaF7HBm3IQxahwAXPl62LDQ==","signature_status":"signed_v1","signed_at":"2026-07-05T05:34:08.951174Z","signed_message":"canonical_sha256_bytes"},"source_id":"2301.07421","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:6414d996aa4bc1fc5c7341f354ee5b7592bf151b98dd8e6f0a08f8e99e20e260","sha256:8bc270029413a39c49a2e1007984f1151142732a5da1f27d44375d2164680770"],"state_sha256":"be17d0bb904e88e8d8e2ac426d3195b1d1bce93579cfeb523df7ae4783e0020c"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"H7Knuzs2GRQvGkfCH2NvVwOiejtTy95LWxSJK/tTHyRRH82QQqZlWyqE1j93KBxO3LUKD/UlBN3BPfppxH46Aw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-01T00:32:06.003548Z","bundle_sha256":"e9b1218319e77f511d79125d8a8002c584e8909e3f54e20f7b90426ca79024d7"}}