{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:H4326EWVO7VNFNBTJTAQTQOK5L","short_pith_number":"pith:H4326EWV","canonical_record":{"source":{"id":"2505.06273","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-06T15:09:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6d1d42080ffae7af697b70902f6e2ec75cfea20d429cd7b29f98728d77c18802","abstract_canon_sha256":"d29a3369da7da910528d5165db848d47c0acf8bae63d4674ef04c74638289a35"},"schema_version":"1.0"},"canonical_sha256":"3f37af12d577ead2b4334cc109c1caeae99b71a55d8f104a6f0402f9c062461d","source":{"kind":"arxiv","id":"2505.06273","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.06273","created_at":"2026-07-05T11:02:24Z"},{"alias_kind":"arxiv_version","alias_value":"2505.06273v2","created_at":"2026-07-05T11:02:24Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.06273","created_at":"2026-07-05T11:02:24Z"},{"alias_kind":"pith_short_12","alias_value":"H4326EWVO7VN","created_at":"2026-07-05T11:02:24Z"},{"alias_kind":"pith_short_16","alias_value":"H4326EWVO7VNFNBT","created_at":"2026-07-05T11:02:24Z"},{"alias_kind":"pith_short_8","alias_value":"H4326EWV","created_at":"2026-07-05T11:02:24Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:H4326EWVO7VNFNBTJTAQTQOK5L","target":"record","payload":{"canonical_record":{"source":{"id":"2505.06273","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-06T15:09:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6d1d42080ffae7af697b70902f6e2ec75cfea20d429cd7b29f98728d77c18802","abstract_canon_sha256":"d29a3369da7da910528d5165db848d47c0acf8bae63d4674ef04c74638289a35"},"schema_version":"1.0"},"canonical_sha256":"3f37af12d577ead2b4334cc109c1caeae99b71a55d8f104a6f0402f9c062461d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:24.761214Z","signature_b64":"jHUBtHV7CVwTP3X+me1GboD6Wj5NBB3Le8nSkBC9L/4YqjEvPJTSGKYoNnhfRG0alzCRQJXnCilu66lcyF1LCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f37af12d577ead2b4334cc109c1caeae99b71a55d8f104a6f0402f9c062461d","last_reissued_at":"2026-07-05T11:02:24.760741Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:24.760741Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.06273","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:02:24Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"sp2uKooAD7szyxthLszgcirv2W65YI2TrQJjZS0uvx1Y8JM7RGheQV9/6qUkccVVrAP0L8JMRNpK/1Z34P4/Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T12:35:42.131484Z"},"content_sha256":"bee2475a2ec618b6d6cbe6a111ada904bc3fa211aed9a288d7d5247d5e2d90b0","schema_version":"1.0","event_id":"sha256:bee2475a2ec618b6d6cbe6a111ada904bc3fa211aed9a288d7d5247d5e2d90b0"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:H4326EWVO7VNFNBTJTAQTQOK5L","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Policy-labeled Preference Learning: Is Preference Enough for RLHF?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dohyeong Kim, Jungwoo Lee, Kyungjae Lee, Seokhun Ju, Seungyub Han, Taehyun Cho","submitted_at":"2025-05-06T15:09:55Z","abstract_excerpt":"To design rewards that align with human goals, Reinforcement Learning from Human Feedback (RLHF) has emerged as a prominent technique for learning reward functions from human preferences and optimizing policies via reinforcement learning algorithms. However, existing RLHF methods often misinterpret trajectories as being generated by an optimal policy, causing inaccurate likelihood estimation and suboptimal learning. Inspired by Direct Preference Optimization framework which directly learns optimal policy without explicit reward, we propose policy-labeled preference learning (PPL), to resolve l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.06273","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.06273/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:02:24Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Pv/079b0zBGqmoIFNh0MUPl3M7YhJvD/7KXJif2w7YEF7rxrSYaKLi/4SK7ocYutGOTBnoUnOwgqbfsIplmdCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T12:35:42.131786Z"},"content_sha256":"a5e73b256b488bacb2786f80586e1b2366dfbb897e8b95595e73ab494877b9d2","schema_version":"1.0","event_id":"sha256:a5e73b256b488bacb2786f80586e1b2366dfbb897e8b95595e73ab494877b9d2"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/H4326EWVO7VNFNBTJTAQTQOK5L/bundle.json","state_url":"https://pith.science/pith/H4326EWVO7VNFNBTJTAQTQOK5L/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/H4326EWVO7VNFNBTJTAQTQOK5L/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T12:35:42Z","links":{"resolver":"https://pith.science/pith/H4326EWVO7VNFNBTJTAQTQOK5L","bundle":"https://pith.science/pith/H4326EWVO7VNFNBTJTAQTQOK5L/bundle.json","state":"https://pith.science/pith/H4326EWVO7VNFNBTJTAQTQOK5L/state.json","well_known_bundle":"https://pith.science/.well-known/pith/H4326EWVO7VNFNBTJTAQTQOK5L/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:H4326EWVO7VNFNBTJTAQTQOK5L","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d29a3369da7da910528d5165db848d47c0acf8bae63d4674ef04c74638289a35","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-06T15:09:55Z","title_canon_sha256":"6d1d42080ffae7af697b70902f6e2ec75cfea20d429cd7b29f98728d77c18802"},"schema_version":"1.0","source":{"id":"2505.06273","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.06273","created_at":"2026-07-05T11:02:24Z"},{"alias_kind":"arxiv_version","alias_value":"2505.06273v2","created_at":"2026-07-05T11:02:24Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.06273","created_at":"2026-07-05T11:02:24Z"},{"alias_kind":"pith_short_12","alias_value":"H4326EWVO7VN","created_at":"2026-07-05T11:02:24Z"},{"alias_kind":"pith_short_16","alias_value":"H4326EWVO7VNFNBT","created_at":"2026-07-05T11:02:24Z"},{"alias_kind":"pith_short_8","alias_value":"H4326EWV","created_at":"2026-07-05T11:02:24Z"}],"graph_snapshots":[{"event_id":"sha256:a5e73b256b488bacb2786f80586e1b2366dfbb897e8b95595e73ab494877b9d2","target":"graph","created_at":"2026-07-05T11:02:24Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.06273/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"To design rewards that align with human goals, Reinforcement Learning from Human Feedback (RLHF) has emerged as a prominent technique for learning reward functions from human preferences and optimizing policies via reinforcement learning algorithms. However, existing RLHF methods often misinterpret trajectories as being generated by an optimal policy, causing inaccurate likelihood estimation and suboptimal learning. Inspired by Direct Preference Optimization framework which directly learns optimal policy without explicit reward, we propose policy-labeled preference learning (PPL), to resolve l","authors_text":"Dohyeong Kim, Jungwoo Lee, Kyungjae Lee, Seokhun Ju, Seungyub Han, Taehyun Cho","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-06T15:09:55Z","title":"Policy-labeled Preference Learning: Is Preference Enough for RLHF?"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.06273","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:bee2475a2ec618b6d6cbe6a111ada904bc3fa211aed9a288d7d5247d5e2d90b0","target":"record","created_at":"2026-07-05T11:02:24Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d29a3369da7da910528d5165db848d47c0acf8bae63d4674ef04c74638289a35","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-06T15:09:55Z","title_canon_sha256":"6d1d42080ffae7af697b70902f6e2ec75cfea20d429cd7b29f98728d77c18802"},"schema_version":"1.0","source":{"id":"2505.06273","kind":"arxiv","version":2}},"canonical_sha256":"3f37af12d577ead2b4334cc109c1caeae99b71a55d8f104a6f0402f9c062461d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3f37af12d577ead2b4334cc109c1caeae99b71a55d8f104a6f0402f9c062461d","first_computed_at":"2026-07-05T11:02:24.760741Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:02:24.760741Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"jHUBtHV7CVwTP3X+me1GboD6Wj5NBB3Le8nSkBC9L/4YqjEvPJTSGKYoNnhfRG0alzCRQJXnCilu66lcyF1LCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:02:24.761214Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.06273","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:bee2475a2ec618b6d6cbe6a111ada904bc3fa211aed9a288d7d5247d5e2d90b0","sha256:a5e73b256b488bacb2786f80586e1b2366dfbb897e8b95595e73ab494877b9d2"],"state_sha256":"238b72c73ce6444cf5fb4c9c8932260bf108e51db1ff97ea3b8b83ca7fa58aeb"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4IDcNSEF7xAMDrftNbSIjfEMU3XWOVXFF0geqMgEsS97ZKFlGfW0KeVk1R6Mm5YcK1TJR/LLZxWkWJ+s7gG+BQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T12:35:42.134314Z","bundle_sha256":"57f770c9668832b9b8faee9e4d4b9d72328903e3da936ee385bcff2240347a92"}}