{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5O4X53VFRZSJN4Z74KL5FC7T4M","short_pith_number":"pith:5O4X53VF","schema_version":"1.0","canonical_sha256":"ebb97eeea58e6496f33fe297d28bf3e310c1b3a3f6af715a4c65320452645e17","source":{"kind":"arxiv","id":"2410.23726","version":1},"attestation_state":"computed","paper":{"title":"Towards Reliable Alignment: Uncertainty-aware RLHF","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Aditya Gopalan, Debangshu Banerjee","submitted_at":"2024-10-31T08:26:51Z","abstract_excerpt":"Recent advances in aligning Large Language Models with human preferences have benefited from larger reward models and better preference data. However, most of these methodologies rely on the accuracy of the reward model. The reward models used in Reinforcement Learning with Human Feedback (RLHF) are typically learned from small datasets using stochastic optimization algorithms, making them prone to high variability. We illustrate the inconsistencies between reward models empirically on numerous open-source datasets.\n  We theoretically show that the fluctuation of the reward models can be detri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.23726","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-31T08:26:51Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a753439bddb5dcd458affc3d8698d3ff2d917fca397f078ce41229fc3991881c","abstract_canon_sha256":"a981823ea79d39074bff2105eca616985f68d748377a2184457f236cf9cc0895"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:29:04.442888Z","signature_b64":"cS3NoPOum5YDsTmCR3Ksi6NHnlMJHgDuy3nbld9E9e11709HgFb8OPPujyMdPep1NJVV35rOWMqOtJbtNrUaDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ebb97eeea58e6496f33fe297d28bf3e310c1b3a3f6af715a4c65320452645e17","last_reissued_at":"2026-07-05T09:29:04.442376Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:29:04.442376Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Reliable Alignment: Uncertainty-aware RLHF","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Aditya Gopalan, Debangshu Banerjee","submitted_at":"2024-10-31T08:26:51Z","abstract_excerpt":"Recent advances in aligning Large Language Models with human preferences have benefited from larger reward models and better preference data. However, most of these methodologies rely on the accuracy of the reward model. The reward models used in Reinforcement Learning with Human Feedback (RLHF) are typically learned from small datasets using stochastic optimization algorithms, making them prone to high variability. We illustrate the inconsistencies between reward models empirically on numerous open-source datasets.\n  We theoretically show that the fluctuation of the reward models can be detri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.23726","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.23726/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.23726","created_at":"2026-07-05T09:29:04.442439+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.23726v1","created_at":"2026-07-05T09:29:04.442439+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.23726","created_at":"2026-07-05T09:29:04.442439+00:00"},{"alias_kind":"pith_short_12","alias_value":"5O4X53VFRZSJ","created_at":"2026-07-05T09:29:04.442439+00:00"},{"alias_kind":"pith_short_16","alias_value":"5O4X53VFRZSJN4Z7","created_at":"2026-07-05T09:29:04.442439+00:00"},{"alias_kind":"pith_short_8","alias_value":"5O4X53VF","created_at":"2026-07-05T09:29:04.442439+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20662","citing_title":"Confidence Laundering in Agent Systems: Why Uncertainty Needs a Latent Carrier","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5O4X53VFRZSJN4Z74KL5FC7T4M","json":"https://pith.science/pith/5O4X53VFRZSJN4Z74KL5FC7T4M.json","graph_json":"https://pith.science/api/pith-number/5O4X53VFRZSJN4Z74KL5FC7T4M/graph.json","events_json":"https://pith.science/api/pith-number/5O4X53VFRZSJN4Z74KL5FC7T4M/events.json","paper":"https://pith.science/paper/5O4X53VF"},"agent_actions":{"view_html":"https://pith.science/pith/5O4X53VFRZSJN4Z74KL5FC7T4M","download_json":"https://pith.science/pith/5O4X53VFRZSJN4Z74KL5FC7T4M.json","view_paper":"https://pith.science/paper/5O4X53VF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.23726&json=true","fetch_graph":"https://pith.science/api/pith-number/5O4X53VFRZSJN4Z74KL5FC7T4M/graph.json","fetch_events":"https://pith.science/api/pith-number/5O4X53VFRZSJN4Z74KL5FC7T4M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5O4X53VFRZSJN4Z74KL5FC7T4M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5O4X53VFRZSJN4Z74KL5FC7T4M/action/storage_attestation","attest_author":"https://pith.science/pith/5O4X53VFRZSJN4Z74KL5FC7T4M/action/author_attestation","sign_citation":"https://pith.science/pith/5O4X53VFRZSJN4Z74KL5FC7T4M/action/citation_signature","submit_replication":"https://pith.science/pith/5O4X53VFRZSJN4Z74KL5FC7T4M/action/replication_record"}},"created_at":"2026-07-05T09:29:04.442439+00:00","updated_at":"2026-07-05T09:29:04.442439+00:00"}