{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:GPLR3XFT2RBA42NEWFSYLSDA75","short_pith_number":"pith:GPLR3XFT","canonical_record":{"source":{"id":"2404.04932","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-07T12:10:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"79d69505175fe3907887adff5c1a501ee3f30342b2affea0ea4923abe1e112f1","abstract_canon_sha256":"372fc1069c71281e62398546eee914713461b697791083951a4a44069d556919"},"schema_version":"1.0"},"canonical_sha256":"33d71ddcb3d4420e69a4b16585c860ff4ae64c3fb12272283990b410247953b7","source":{"kind":"arxiv","id":"2404.04932","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.04932","created_at":"2026-07-05T08:05:23Z"},{"alias_kind":"arxiv_version","alias_value":"2404.04932v1","created_at":"2026-07-05T08:05:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.04932","created_at":"2026-07-05T08:05:23Z"},{"alias_kind":"pith_short_12","alias_value":"GPLR3XFT2RBA","created_at":"2026-07-05T08:05:23Z"},{"alias_kind":"pith_short_16","alias_value":"GPLR3XFT2RBA42NE","created_at":"2026-07-05T08:05:23Z"},{"alias_kind":"pith_short_8","alias_value":"GPLR3XFT","created_at":"2026-07-05T08:05:23Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:GPLR3XFT2RBA42NEWFSYLSDA75","target":"record","payload":{"canonical_record":{"source":{"id":"2404.04932","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-07T12:10:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"79d69505175fe3907887adff5c1a501ee3f30342b2affea0ea4923abe1e112f1","abstract_canon_sha256":"372fc1069c71281e62398546eee914713461b697791083951a4a44069d556919"},"schema_version":"1.0"},"canonical_sha256":"33d71ddcb3d4420e69a4b16585c860ff4ae64c3fb12272283990b410247953b7","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:23.741930Z","signature_b64":"PWwshmqJROD0BkbFiYlGPpS0mqGuIcO+bkPqJJOsjKjuRSlJZXFFM5SxFcWOL82HhDEXMEnVsOvLxk3XMmqxCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"33d71ddcb3d4420e69a4b16585c860ff4ae64c3fb12272283990b410247953b7","last_reissued_at":"2026-07-05T08:05:23.741438Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:23.741438Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2404.04932","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:05:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"QdL4pcG6IBgdigBiWZAtBNd0Ppb3kNnKLnbzYW1DPcRJqn6oslEeM+xtR/MEVOfAAp0MGqVdioGnluyl2qvYCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T02:50:15.762906Z"},"content_sha256":"b6484d4e329de20d3a933c201d1b8980bde35176ab290e7d34b30d3c9e11d21f","schema_version":"1.0","event_id":"sha256:b6484d4e329de20d3a933c201d1b8980bde35176ab290e7d34b30d3c9e11d21f"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:GPLR3XFT2RBA42NEWFSYLSDA75","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Towards Understanding the Influence of Reward Margin on Preference Model Performance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bowen Qin, Duanyu Feng, Xi Yang","submitted_at":"2024-04-07T12:10:04Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is a widely used framework for the training of language models. However, the process of using RLHF to develop a language model that is well-aligned presents challenges, especially when it comes to optimizing the reward model. Our research has found that existing reward models, when trained using the traditional ranking objective based on human preference data, often struggle to effectively distinguish between responses that are more or less favorable in real-world scenarios. To bridge this gap, our study introduces a novel method to estimate th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.04932","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.04932/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:05:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"7ijckLnuatQDjTYY5H/kwCv+lcL5897pHZSFm4pXKZLZOOFJN6tuyOdz2YJB5DfPdjrhU99XWpXbzNnKR0Z8Cg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T02:50:15.763844Z"},"content_sha256":"cab3049b477e7192253e874f7a283b3753701c05e15c9c2b9f3363a4e631c934","schema_version":"1.0","event_id":"sha256:cab3049b477e7192253e874f7a283b3753701c05e15c9c2b9f3363a4e631c934"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GPLR3XFT2RBA42NEWFSYLSDA75/bundle.json","state_url":"https://pith.science/pith/GPLR3XFT2RBA42NEWFSYLSDA75/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GPLR3XFT2RBA42NEWFSYLSDA75/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T02:50:15Z","links":{"resolver":"https://pith.science/pith/GPLR3XFT2RBA42NEWFSYLSDA75","bundle":"https://pith.science/pith/GPLR3XFT2RBA42NEWFSYLSDA75/bundle.json","state":"https://pith.science/pith/GPLR3XFT2RBA42NEWFSYLSDA75/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GPLR3XFT2RBA42NEWFSYLSDA75/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:GPLR3XFT2RBA42NEWFSYLSDA75","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"372fc1069c71281e62398546eee914713461b697791083951a4a44069d556919","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-07T12:10:04Z","title_canon_sha256":"79d69505175fe3907887adff5c1a501ee3f30342b2affea0ea4923abe1e112f1"},"schema_version":"1.0","source":{"id":"2404.04932","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.04932","created_at":"2026-07-05T08:05:23Z"},{"alias_kind":"arxiv_version","alias_value":"2404.04932v1","created_at":"2026-07-05T08:05:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.04932","created_at":"2026-07-05T08:05:23Z"},{"alias_kind":"pith_short_12","alias_value":"GPLR3XFT2RBA","created_at":"2026-07-05T08:05:23Z"},{"alias_kind":"pith_short_16","alias_value":"GPLR3XFT2RBA42NE","created_at":"2026-07-05T08:05:23Z"},{"alias_kind":"pith_short_8","alias_value":"GPLR3XFT","created_at":"2026-07-05T08:05:23Z"}],"graph_snapshots":[{"event_id":"sha256:cab3049b477e7192253e874f7a283b3753701c05e15c9c2b9f3363a4e631c934","target":"graph","created_at":"2026-07-05T08:05:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2404.04932/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is a widely used framework for the training of language models. However, the process of using RLHF to develop a language model that is well-aligned presents challenges, especially when it comes to optimizing the reward model. Our research has found that existing reward models, when trained using the traditional ranking objective based on human preference data, often struggle to effectively distinguish between responses that are more or less favorable in real-world scenarios. To bridge this gap, our study introduces a novel method to estimate th","authors_text":"Bowen Qin, Duanyu Feng, Xi Yang","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-07T12:10:04Z","title":"Towards Understanding the Influence of Reward Margin on Preference Model Performance"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.04932","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b6484d4e329de20d3a933c201d1b8980bde35176ab290e7d34b30d3c9e11d21f","target":"record","created_at":"2026-07-05T08:05:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"372fc1069c71281e62398546eee914713461b697791083951a4a44069d556919","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-07T12:10:04Z","title_canon_sha256":"79d69505175fe3907887adff5c1a501ee3f30342b2affea0ea4923abe1e112f1"},"schema_version":"1.0","source":{"id":"2404.04932","kind":"arxiv","version":1}},"canonical_sha256":"33d71ddcb3d4420e69a4b16585c860ff4ae64c3fb12272283990b410247953b7","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"33d71ddcb3d4420e69a4b16585c860ff4ae64c3fb12272283990b410247953b7","first_computed_at":"2026-07-05T08:05:23.741438Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:05:23.741438Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"PWwshmqJROD0BkbFiYlGPpS0mqGuIcO+bkPqJJOsjKjuRSlJZXFFM5SxFcWOL82HhDEXMEnVsOvLxk3XMmqxCw==","signature_status":"signed_v1","signed_at":"2026-07-05T08:05:23.741930Z","signed_message":"canonical_sha256_bytes"},"source_id":"2404.04932","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b6484d4e329de20d3a933c201d1b8980bde35176ab290e7d34b30d3c9e11d21f","sha256:cab3049b477e7192253e874f7a283b3753701c05e15c9c2b9f3363a4e631c934"],"state_sha256":"d404327622f58a7a9576b7a0619499a05abbb41ea0a2e0f06f4a14b1bccf0a21"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"YTXSMV5WjDM1vRBi7TqSdglenEk362CB/aX9VvrQFOx71wIgzEs5TmfjcACtIxRVT9Z+otIZcn9u5ITUWqwlCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T02:50:15.770025Z","bundle_sha256":"68f3b96c05ef88dfa41712c548d79eaba2df5c45e46951febc0a6e3f56925823"}}