{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:4LNLSPMPR5CQXP3NUBW6IS4I4W","short_pith_number":"pith:4LNLSPMP","canonical_record":{"source":{"id":"2411.00418","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-01T07:29:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"305b6206a8c60180f123c2f2f34b8500eba728540fcc6ae9b82dc523810fc65c","abstract_canon_sha256":"cff8395948e0cc5539389c15cce61ea78b73327e7e689c60d6b43b15478df93d"},"schema_version":"1.0"},"canonical_sha256":"e2dab93d8f8f450bbf6da06de44b88e593803bdb94e21b4df0fe3b20345efb71","source":{"kind":"arxiv","id":"2411.00418","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.00418","created_at":"2026-07-05T11:14:50Z"},{"alias_kind":"arxiv_version","alias_value":"2411.00418v3","created_at":"2026-07-05T11:14:50Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.00418","created_at":"2026-07-05T11:14:50Z"},{"alias_kind":"pith_short_12","alias_value":"4LNLSPMPR5CQ","created_at":"2026-07-05T11:14:50Z"},{"alias_kind":"pith_short_16","alias_value":"4LNLSPMPR5CQXP3N","created_at":"2026-07-05T11:14:50Z"},{"alias_kind":"pith_short_8","alias_value":"4LNLSPMP","created_at":"2026-07-05T11:14:50Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:4LNLSPMPR5CQXP3NUBW6IS4I4W","target":"record","payload":{"canonical_record":{"source":{"id":"2411.00418","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-01T07:29:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"305b6206a8c60180f123c2f2f34b8500eba728540fcc6ae9b82dc523810fc65c","abstract_canon_sha256":"cff8395948e0cc5539389c15cce61ea78b73327e7e689c60d6b43b15478df93d"},"schema_version":"1.0"},"canonical_sha256":"e2dab93d8f8f450bbf6da06de44b88e593803bdb94e21b4df0fe3b20345efb71","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:50.159635Z","signature_b64":"l6F/tUn+nvDJNYvz+G+wAKWZ1ZO3osQ4h18J8EvvbjWRL98aXsYuBJSYR+U8D/Kf8pMPsKEShL7tGb5Ei7WCBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e2dab93d8f8f450bbf6da06de44b88e593803bdb94e21b4df0fe3b20345efb71","last_reissued_at":"2026-07-05T11:14:50.159162Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:50.159162Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2411.00418","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:14:50Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"saE/9+zzgk5SY56bVVdcLzd9ypxKl20dFY+AjMMPJ/Qtltnx2g75kQGwd04YHh0sQG9uyoneM2d05S+t1SgpDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:45:12.717180Z"},"content_sha256":"da80f7a8c13596278e1631d87e69b3bc2fccf29af39ec0f37ee64e0d3b3f9a33","schema_version":"1.0","event_id":"sha256:da80f7a8c13596278e1631d87e69b3bc2fccf29af39ec0f37ee64e0d3b3f9a33"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:4LNLSPMPR5CQXP3NUBW6IS4I4W","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Self-Evolved Reward Learning for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chenghua Huang, Dongmei Zhang, Fangkai Yang, Lu Wang, Pu Zhao, Qingwei Lin, Qi Zhang, Saravan Rajmohan, Zeqi Lin, Zhizhen Fan","submitted_at":"2024-11-01T07:29:03Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is a crucial technique for aligning language models with human preferences, playing a pivotal role in the success of conversational models like GPT-4, ChatGPT, and Llama 2. A core challenge in employing RLHF lies in training a reliable reward model (RM), which relies on high-quality labels typically provided by human experts or advanced AI system. These methods can be costly and may introduce biases that affect the language model's responses. As language models improve, human input may become less effective in further enhancing their performanc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.00418","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.00418/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:14:50Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bNWolZtzWaEtLBJjkq5PHPZevrqBBt21uaH8fSHtNibtIhdx+3IL7OmCAyOv5KUG8GBGsugtJ9b0F0RZQzNYAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:45:12.718004Z"},"content_sha256":"128ead91aa1489818560379aaa41fac397a56b8445306462ad4072ab0c473540","schema_version":"1.0","event_id":"sha256:128ead91aa1489818560379aaa41fac397a56b8445306462ad4072ab0c473540"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/4LNLSPMPR5CQXP3NUBW6IS4I4W/bundle.json","state_url":"https://pith.science/pith/4LNLSPMPR5CQXP3NUBW6IS4I4W/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/4LNLSPMPR5CQXP3NUBW6IS4I4W/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T07:45:12Z","links":{"resolver":"https://pith.science/pith/4LNLSPMPR5CQXP3NUBW6IS4I4W","bundle":"https://pith.science/pith/4LNLSPMPR5CQXP3NUBW6IS4I4W/bundle.json","state":"https://pith.science/pith/4LNLSPMPR5CQXP3NUBW6IS4I4W/state.json","well_known_bundle":"https://pith.science/.well-known/pith/4LNLSPMPR5CQXP3NUBW6IS4I4W/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:4LNLSPMPR5CQXP3NUBW6IS4I4W","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"cff8395948e0cc5539389c15cce61ea78b73327e7e689c60d6b43b15478df93d","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-01T07:29:03Z","title_canon_sha256":"305b6206a8c60180f123c2f2f34b8500eba728540fcc6ae9b82dc523810fc65c"},"schema_version":"1.0","source":{"id":"2411.00418","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.00418","created_at":"2026-07-05T11:14:50Z"},{"alias_kind":"arxiv_version","alias_value":"2411.00418v3","created_at":"2026-07-05T11:14:50Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.00418","created_at":"2026-07-05T11:14:50Z"},{"alias_kind":"pith_short_12","alias_value":"4LNLSPMPR5CQ","created_at":"2026-07-05T11:14:50Z"},{"alias_kind":"pith_short_16","alias_value":"4LNLSPMPR5CQXP3N","created_at":"2026-07-05T11:14:50Z"},{"alias_kind":"pith_short_8","alias_value":"4LNLSPMP","created_at":"2026-07-05T11:14:50Z"}],"graph_snapshots":[{"event_id":"sha256:128ead91aa1489818560379aaa41fac397a56b8445306462ad4072ab0c473540","target":"graph","created_at":"2026-07-05T11:14:50Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2411.00418/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is a crucial technique for aligning language models with human preferences, playing a pivotal role in the success of conversational models like GPT-4, ChatGPT, and Llama 2. A core challenge in employing RLHF lies in training a reliable reward model (RM), which relies on high-quality labels typically provided by human experts or advanced AI system. These methods can be costly and may introduce biases that affect the language model's responses. As language models improve, human input may become less effective in further enhancing their performanc","authors_text":"Chenghua Huang, Dongmei Zhang, Fangkai Yang, Lu Wang, Pu Zhao, Qingwei Lin, Qi Zhang, Saravan Rajmohan, Zeqi Lin, Zhizhen Fan","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-01T07:29:03Z","title":"Self-Evolved Reward Learning for LLMs"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.00418","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:da80f7a8c13596278e1631d87e69b3bc2fccf29af39ec0f37ee64e0d3b3f9a33","target":"record","created_at":"2026-07-05T11:14:50Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"cff8395948e0cc5539389c15cce61ea78b73327e7e689c60d6b43b15478df93d","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-01T07:29:03Z","title_canon_sha256":"305b6206a8c60180f123c2f2f34b8500eba728540fcc6ae9b82dc523810fc65c"},"schema_version":"1.0","source":{"id":"2411.00418","kind":"arxiv","version":3}},"canonical_sha256":"e2dab93d8f8f450bbf6da06de44b88e593803bdb94e21b4df0fe3b20345efb71","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e2dab93d8f8f450bbf6da06de44b88e593803bdb94e21b4df0fe3b20345efb71","first_computed_at":"2026-07-05T11:14:50.159162Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:14:50.159162Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"l6F/tUn+nvDJNYvz+G+wAKWZ1ZO3osQ4h18J8EvvbjWRL98aXsYuBJSYR+U8D/Kf8pMPsKEShL7tGb5Ei7WCBw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:14:50.159635Z","signed_message":"canonical_sha256_bytes"},"source_id":"2411.00418","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:da80f7a8c13596278e1631d87e69b3bc2fccf29af39ec0f37ee64e0d3b3f9a33","sha256:128ead91aa1489818560379aaa41fac397a56b8445306462ad4072ab0c473540"],"state_sha256":"f11a5d04fccf5d7a66a241fcaa918e8fdb7d0d690ec3dda9a219ba9e4e3f221f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"RYMSTRq4D0ZIyiC78nZh8Vdka8jjWZvqR2RJC199HPptZICfqrf/1KVh00BW+h89r8j2x/OKcyhqeXPFTFpICA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T07:45:12.843338Z","bundle_sha256":"7652e3c5edd416b4320a2616fe82eba1e3cbee715ec59378d75be14d6eb64cb4"}}