{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:V6EKFS4YSW3TOBUIOP6LFROA63","short_pith_number":"pith:V6EKFS4Y","canonical_record":{"source":{"id":"2402.14228","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-22T02:20:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ecd340bd30381f441a61f7e0a0c8f7b5069792cde48fa87a4fe62edcb0f76868","abstract_canon_sha256":"7c527976e229f2f30c6f402eb5c61e9978aad256bd092f789d544f1beb6f1ebc"},"schema_version":"1.0"},"canonical_sha256":"af88a2cb9895b737068873fcb2c5c0f6f51708b2c3243d0f1e544c2aeeeb1f1f","source":{"kind":"arxiv","id":"2402.14228","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2402.14228","created_at":"2026-07-05T09:52:37Z"},{"alias_kind":"arxiv_version","alias_value":"2402.14228v3","created_at":"2026-07-05T09:52:37Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14228","created_at":"2026-07-05T09:52:37Z"},{"alias_kind":"pith_short_12","alias_value":"V6EKFS4YSW3T","created_at":"2026-07-05T09:52:37Z"},{"alias_kind":"pith_short_16","alias_value":"V6EKFS4YSW3TOBUI","created_at":"2026-07-05T09:52:37Z"},{"alias_kind":"pith_short_8","alias_value":"V6EKFS4Y","created_at":"2026-07-05T09:52:37Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:V6EKFS4YSW3TOBUIOP6LFROA63","target":"record","payload":{"canonical_record":{"source":{"id":"2402.14228","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-22T02:20:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ecd340bd30381f441a61f7e0a0c8f7b5069792cde48fa87a4fe62edcb0f76868","abstract_canon_sha256":"7c527976e229f2f30c6f402eb5c61e9978aad256bd092f789d544f1beb6f1ebc"},"schema_version":"1.0"},"canonical_sha256":"af88a2cb9895b737068873fcb2c5c0f6f51708b2c3243d0f1e544c2aeeeb1f1f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:37.598575Z","signature_b64":"3cYk3xypVG+UfwRdYYOREyRkjvXTOra3nvB+3NesVPn0B4ZQkkaFusK00vB62QcEMtcrOW5B74WYtdazFojwBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af88a2cb9895b737068873fcb2c5c0f6f51708b2c3243d0f1e544c2aeeeb1f1f","last_reissued_at":"2026-07-05T09:52:37.597991Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:37.597991Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2402.14228","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:52:37Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"L7KSD9N7C44NbRBFIVDo0bpYJ1NGF1PPyC8LaP2Z4Kk7jZCUrkGniRZMGX4pC05t3qgNEl+aP2avyInMUF1NBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T00:52:57.181034Z"},"content_sha256":"6641e0f6e3d496bfbe17e79b6c605818ef04d5bbfdcf6aef458bfa7ec83b3b06","schema_version":"1.0","event_id":"sha256:6641e0f6e3d496bfbe17e79b6c605818ef04d5bbfdcf6aef458bfa7ec83b3b06"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:V6EKFS4YSW3TOBUIOP6LFROA63","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"COPR: Continual Human Preference Learning via Optimal Policy Regularization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bin Liang, Han Zhang, Hui Wang, Kam-Fai Wong, Lin Gui, Ruifeng Xu, Yehong Zhang, Yuanzhao Zhai, Yue Yu, Yulan He, Yu Lei","submitted_at":"2024-02-22T02:20:08Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is commonly utilized to improve the alignment of Large Language Models (LLMs) with human preferences. Given the evolving nature of human preferences, continual alignment becomes more crucial and practical in comparison to traditional static alignment. Nevertheless, making RLHF compatible with Continual Learning (CL) is challenging due to its complex process. Meanwhile, directly learning new human preferences may lead to Catastrophic Forgetting (CF) of historical preferences, resulting in helpless or harmful outputs. To overcome these challenges"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14228","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14228/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:52:37Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Z8KYCyBuA9wwgZujDtytRSQ0k7u3Rrqaul1C8YKJxIJoQXNjhQehYo+g4/kNC6O1/2AL4c1thKIJfU22tOLbAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T00:52:57.181566Z"},"content_sha256":"d79dfdb84f19889fe04cfc723e7686f02310f73fdd1c257d7d6f6f2cddf07241","schema_version":"1.0","event_id":"sha256:d79dfdb84f19889fe04cfc723e7686f02310f73fdd1c257d7d6f6f2cddf07241"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/V6EKFS4YSW3TOBUIOP6LFROA63/bundle.json","state_url":"https://pith.science/pith/V6EKFS4YSW3TOBUIOP6LFROA63/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/V6EKFS4YSW3TOBUIOP6LFROA63/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T00:52:57Z","links":{"resolver":"https://pith.science/pith/V6EKFS4YSW3TOBUIOP6LFROA63","bundle":"https://pith.science/pith/V6EKFS4YSW3TOBUIOP6LFROA63/bundle.json","state":"https://pith.science/pith/V6EKFS4YSW3TOBUIOP6LFROA63/state.json","well_known_bundle":"https://pith.science/.well-known/pith/V6EKFS4YSW3TOBUIOP6LFROA63/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:V6EKFS4YSW3TOBUIOP6LFROA63","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"7c527976e229f2f30c6f402eb5c61e9978aad256bd092f789d544f1beb6f1ebc","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-22T02:20:08Z","title_canon_sha256":"ecd340bd30381f441a61f7e0a0c8f7b5069792cde48fa87a4fe62edcb0f76868"},"schema_version":"1.0","source":{"id":"2402.14228","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2402.14228","created_at":"2026-07-05T09:52:37Z"},{"alias_kind":"arxiv_version","alias_value":"2402.14228v3","created_at":"2026-07-05T09:52:37Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14228","created_at":"2026-07-05T09:52:37Z"},{"alias_kind":"pith_short_12","alias_value":"V6EKFS4YSW3T","created_at":"2026-07-05T09:52:37Z"},{"alias_kind":"pith_short_16","alias_value":"V6EKFS4YSW3TOBUI","created_at":"2026-07-05T09:52:37Z"},{"alias_kind":"pith_short_8","alias_value":"V6EKFS4Y","created_at":"2026-07-05T09:52:37Z"}],"graph_snapshots":[{"event_id":"sha256:d79dfdb84f19889fe04cfc723e7686f02310f73fdd1c257d7d6f6f2cddf07241","target":"graph","created_at":"2026-07-05T09:52:37Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2402.14228/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is commonly utilized to improve the alignment of Large Language Models (LLMs) with human preferences. Given the evolving nature of human preferences, continual alignment becomes more crucial and practical in comparison to traditional static alignment. Nevertheless, making RLHF compatible with Continual Learning (CL) is challenging due to its complex process. Meanwhile, directly learning new human preferences may lead to Catastrophic Forgetting (CF) of historical preferences, resulting in helpless or harmful outputs. To overcome these challenges","authors_text":"Bin Liang, Han Zhang, Hui Wang, Kam-Fai Wong, Lin Gui, Ruifeng Xu, Yehong Zhang, Yuanzhao Zhai, Yue Yu, Yulan He, Yu Lei","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-22T02:20:08Z","title":"COPR: Continual Human Preference Learning via Optimal Policy Regularization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14228","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:6641e0f6e3d496bfbe17e79b6c605818ef04d5bbfdcf6aef458bfa7ec83b3b06","target":"record","created_at":"2026-07-05T09:52:37Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"7c527976e229f2f30c6f402eb5c61e9978aad256bd092f789d544f1beb6f1ebc","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-22T02:20:08Z","title_canon_sha256":"ecd340bd30381f441a61f7e0a0c8f7b5069792cde48fa87a4fe62edcb0f76868"},"schema_version":"1.0","source":{"id":"2402.14228","kind":"arxiv","version":3}},"canonical_sha256":"af88a2cb9895b737068873fcb2c5c0f6f51708b2c3243d0f1e544c2aeeeb1f1f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"af88a2cb9895b737068873fcb2c5c0f6f51708b2c3243d0f1e544c2aeeeb1f1f","first_computed_at":"2026-07-05T09:52:37.597991Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:52:37.597991Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"3cYk3xypVG+UfwRdYYOREyRkjvXTOra3nvB+3NesVPn0B4ZQkkaFusK00vB62QcEMtcrOW5B74WYtdazFojwBw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:52:37.598575Z","signed_message":"canonical_sha256_bytes"},"source_id":"2402.14228","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:6641e0f6e3d496bfbe17e79b6c605818ef04d5bbfdcf6aef458bfa7ec83b3b06","sha256:d79dfdb84f19889fe04cfc723e7686f02310f73fdd1c257d7d6f6f2cddf07241"],"state_sha256":"d075cccb918a9f1c7b0321f3afc7b0821b3314782dcd06f57169343ac1111b8d"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8hnQET+k/jn25M6hFy+OD3pOOOxm4tRZz4omMupGSF0Y49+vfgi9xiJWEqMx+sy0Z7JJyyENpg1FX+xOHRS0AQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T00:52:57.185989Z","bundle_sha256":"411d7a9cb2b7125b59213d97b086103a90fbb605edc34618c46d29aed75f2b4e"}}