{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:KJQPR4JOGFT45AYTI77SPM3MGI","short_pith_number":"pith:KJQPR4JO","canonical_record":{"source":{"id":"2504.14732","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-20T20:09:19Z","cross_cats_sorted":[],"title_canon_sha256":"58dafa5d44f0324de7a5fd5a99ed6c812a1f8ff82849edb64c9f28f948508e9f","abstract_canon_sha256":"b4efced2a1b0c83675570c8bca3a87c4c091c7c5d0e432c116b13d8863dab7b0"},"schema_version":"1.0"},"canonical_sha256":"5260f8f12e3167ce831347ff27b36c323a58d54a58b033702de0bf06b0c2db7a","source":{"kind":"arxiv","id":"2504.14732","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.14732","created_at":"2026-07-05T10:54:16Z"},{"alias_kind":"arxiv_version","alias_value":"2504.14732v3","created_at":"2026-07-05T10:54:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14732","created_at":"2026-07-05T10:54:16Z"},{"alias_kind":"pith_short_12","alias_value":"KJQPR4JOGFT4","created_at":"2026-07-05T10:54:16Z"},{"alias_kind":"pith_short_16","alias_value":"KJQPR4JOGFT45AYT","created_at":"2026-07-05T10:54:16Z"},{"alias_kind":"pith_short_8","alias_value":"KJQPR4JO","created_at":"2026-07-05T10:54:16Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:KJQPR4JOGFT45AYTI77SPM3MGI","target":"record","payload":{"canonical_record":{"source":{"id":"2504.14732","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-20T20:09:19Z","cross_cats_sorted":[],"title_canon_sha256":"58dafa5d44f0324de7a5fd5a99ed6c812a1f8ff82849edb64c9f28f948508e9f","abstract_canon_sha256":"b4efced2a1b0c83675570c8bca3a87c4c091c7c5d0e432c116b13d8863dab7b0"},"schema_version":"1.0"},"canonical_sha256":"5260f8f12e3167ce831347ff27b36c323a58d54a58b033702de0bf06b0c2db7a","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:16.508639Z","signature_b64":"6QLGADCr3lp50ykH2LcN2W4tWgspZ7ohRjvytiUeu3QbOyfRqRQMH5o+MvtXpGea9wNBNLr5OpoqkBo0qO4SBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5260f8f12e3167ce831347ff27b36c323a58d54a58b033702de0bf06b0c2db7a","last_reissued_at":"2026-07-05T10:54:16.508168Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:16.508168Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2504.14732","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:54:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"98Oatzr1/6yHT3ps25bt23dji3+cZTbp/6wCWNQTndOQummTSszBauz1dh7yNuEYpqtTq4dBiuWGnSOE6a/xDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T13:25:07.849904Z"},"content_sha256":"d2f5da2ed06aa6676406373112bf3cc55e60505eae774c1ad210436b4b08dd95","schema_version":"1.0","event_id":"sha256:d2f5da2ed06aa6676406373112bf3cc55e60505eae774c1ad210436b4b08dd95"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:KJQPR4JOGFT45AYTI77SPM3MGI","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reinforcement Learning from Multi-level and Episodic Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Maheed H. Ahmed, Mahsa Ghasemi, Muhammad Qasim Elahi, Somtochukwu Oguchienti","submitted_at":"2025-04-20T20:09:19Z","abstract_excerpt":"Designing an effective reward function has long been a challenge in reinforcement learning, particularly for complex tasks in unstructured environments. To address this, various learning paradigms have emerged that leverage different forms of human input to specify or refine the reward function. Reinforcement learning from human feedback is a prominent approach that utilizes human comparative feedback, expressed as a preference for one behavior over another, to tackle this problem. In contrast to comparative feedback, we explore multi-level human feedback, which is provided in the form of a sc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14732","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.14732/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:54:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"d9zmMWm9V3PJWhm8cm5r1mmCLEfHpnIhEPkQRF+x9mzDcUv6FwPd6R3nE/ZJEb3GlMOI/UxT6XxODiYvlcs1Dg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T13:25:07.850806Z"},"content_sha256":"3b2f4e7e903965d57fd81be94d45bb98e76ca18b344c7c5e6b9d5746ba7886af","schema_version":"1.0","event_id":"sha256:3b2f4e7e903965d57fd81be94d45bb98e76ca18b344c7c5e6b9d5746ba7886af"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/KJQPR4JOGFT45AYTI77SPM3MGI/bundle.json","state_url":"https://pith.science/pith/KJQPR4JOGFT45AYTI77SPM3MGI/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/KJQPR4JOGFT45AYTI77SPM3MGI/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T13:25:07Z","links":{"resolver":"https://pith.science/pith/KJQPR4JOGFT45AYTI77SPM3MGI","bundle":"https://pith.science/pith/KJQPR4JOGFT45AYTI77SPM3MGI/bundle.json","state":"https://pith.science/pith/KJQPR4JOGFT45AYTI77SPM3MGI/state.json","well_known_bundle":"https://pith.science/.well-known/pith/KJQPR4JOGFT45AYTI77SPM3MGI/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:KJQPR4JOGFT45AYTI77SPM3MGI","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b4efced2a1b0c83675570c8bca3a87c4c091c7c5d0e432c116b13d8863dab7b0","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-20T20:09:19Z","title_canon_sha256":"58dafa5d44f0324de7a5fd5a99ed6c812a1f8ff82849edb64c9f28f948508e9f"},"schema_version":"1.0","source":{"id":"2504.14732","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.14732","created_at":"2026-07-05T10:54:16Z"},{"alias_kind":"arxiv_version","alias_value":"2504.14732v3","created_at":"2026-07-05T10:54:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14732","created_at":"2026-07-05T10:54:16Z"},{"alias_kind":"pith_short_12","alias_value":"KJQPR4JOGFT4","created_at":"2026-07-05T10:54:16Z"},{"alias_kind":"pith_short_16","alias_value":"KJQPR4JOGFT45AYT","created_at":"2026-07-05T10:54:16Z"},{"alias_kind":"pith_short_8","alias_value":"KJQPR4JO","created_at":"2026-07-05T10:54:16Z"}],"graph_snapshots":[{"event_id":"sha256:3b2f4e7e903965d57fd81be94d45bb98e76ca18b344c7c5e6b9d5746ba7886af","target":"graph","created_at":"2026-07-05T10:54:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2504.14732/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Designing an effective reward function has long been a challenge in reinforcement learning, particularly for complex tasks in unstructured environments. To address this, various learning paradigms have emerged that leverage different forms of human input to specify or refine the reward function. Reinforcement learning from human feedback is a prominent approach that utilizes human comparative feedback, expressed as a preference for one behavior over another, to tackle this problem. In contrast to comparative feedback, we explore multi-level human feedback, which is provided in the form of a sc","authors_text":"Maheed H. Ahmed, Mahsa Ghasemi, Muhammad Qasim Elahi, Somtochukwu Oguchienti","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-20T20:09:19Z","title":"Reinforcement Learning from Multi-level and Episodic Human Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14732","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d2f5da2ed06aa6676406373112bf3cc55e60505eae774c1ad210436b4b08dd95","target":"record","created_at":"2026-07-05T10:54:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b4efced2a1b0c83675570c8bca3a87c4c091c7c5d0e432c116b13d8863dab7b0","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-20T20:09:19Z","title_canon_sha256":"58dafa5d44f0324de7a5fd5a99ed6c812a1f8ff82849edb64c9f28f948508e9f"},"schema_version":"1.0","source":{"id":"2504.14732","kind":"arxiv","version":3}},"canonical_sha256":"5260f8f12e3167ce831347ff27b36c323a58d54a58b033702de0bf06b0c2db7a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"5260f8f12e3167ce831347ff27b36c323a58d54a58b033702de0bf06b0c2db7a","first_computed_at":"2026-07-05T10:54:16.508168Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:54:16.508168Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"6QLGADCr3lp50ykH2LcN2W4tWgspZ7ohRjvytiUeu3QbOyfRqRQMH5o+MvtXpGea9wNBNLr5OpoqkBo0qO4SBg==","signature_status":"signed_v1","signed_at":"2026-07-05T10:54:16.508639Z","signed_message":"canonical_sha256_bytes"},"source_id":"2504.14732","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d2f5da2ed06aa6676406373112bf3cc55e60505eae774c1ad210436b4b08dd95","sha256:3b2f4e7e903965d57fd81be94d45bb98e76ca18b344c7c5e6b9d5746ba7886af"],"state_sha256":"d56b0f2b5ff853d8d5a2d936c12463f901514b9bad362d3374c0a6ec59199ab8"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CYQGfB4p4xXB1kVD8BiyFkBW94neVKh9SDkfl7CRbVfcxCgAnRxJ8jOqUp82RmJMMekdQLrkLivXlEMV6fyVAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T13:25:07.974652Z","bundle_sha256":"51af08844972953db3116e23837ae8485118acb88ebe43b04b99a81564ac50cc"}}