{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QN3ABX2IIPHFIXAKK2GJAU6ZUP","short_pith_number":"pith:QN3ABX2I","schema_version":"1.0","canonical_sha256":"837600df4843ce545c0a568c9053d9a3d34155a5cc1dbb01d58780b4b3ed737b","source":{"kind":"arxiv","id":"2410.01257","version":2},"attestation_state":"computed","paper":{"title":"HelpSteer2-Preference: Complementing Ratings with Preferences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alexander Bukharin, Daniel Egert, Gerald Shen, Jiaqi Zeng, Oleksii Kuchaiev, Olivier Delalleau, Yi Dong, Zhilin Wang","submitted_at":"2024-10-02T06:05:52Z","abstract_excerpt":"Reward models are critical for aligning models to follow instructions, and are typically trained following one of two popular paradigms: Bradley-Terry style or Regression style. However, there is a lack of evidence that either approach is better than the other, when adequately matched for data. This is primarily because these approaches require data collected in different (but incompatible) formats, meaning that adequately matched data is not available in existing public datasets. To tackle this problem, we release preference annotations (designed for Bradley-Terry training) to complement exis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01257","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T06:05:52Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"ba40e4fca8e5350dba90f36b358698ac3efe2ce8ef2396a0df24c82cd3fa7d6a","abstract_canon_sha256":"152fc2265e177d7adb23f1934dd9d9848fe3ea18996c17361b711a4314edc27a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:10.900345Z","signature_b64":"7z+N6RMoankn0GOdF5oHKsGCo8fitU6GOHMnqFW1bXbfufnteyCY7SQc/vuBDKZYzjvwWXtptoD8RJtMDJtvDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"837600df4843ce545c0a568c9053d9a3d34155a5cc1dbb01d58780b4b3ed737b","last_reissued_at":"2026-07-05T10:25:10.899736Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:10.899736Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HelpSteer2-Preference: Complementing Ratings with Preferences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alexander Bukharin, Daniel Egert, Gerald Shen, Jiaqi Zeng, Oleksii Kuchaiev, Olivier Delalleau, Yi Dong, Zhilin Wang","submitted_at":"2024-10-02T06:05:52Z","abstract_excerpt":"Reward models are critical for aligning models to follow instructions, and are typically trained following one of two popular paradigms: Bradley-Terry style or Regression style. However, there is a lack of evidence that either approach is better than the other, when adequately matched for data. This is primarily because these approaches require data collected in different (but incompatible) formats, meaning that adequately matched data is not available in existing public datasets. To tackle this problem, we release preference annotations (designed for Bradley-Terry training) to complement exis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01257","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01257/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01257","created_at":"2026-07-05T10:25:10.899806+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01257v2","created_at":"2026-07-05T10:25:10.899806+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01257","created_at":"2026-07-05T10:25:10.899806+00:00"},{"alias_kind":"pith_short_12","alias_value":"QN3ABX2IIPHF","created_at":"2026-07-05T10:25:10.899806+00:00"},{"alias_kind":"pith_short_16","alias_value":"QN3ABX2IIPHFIXAK","created_at":"2026-07-05T10:25:10.899806+00:00"},{"alias_kind":"pith_short_8","alias_value":"QN3ABX2I","created_at":"2026-07-05T10:25:10.899806+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01937","citing_title":"RewardBench 2: Advancing Reward Model Evaluation","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2506.10060","citing_title":"Textual Bayes: Quantifying Prompt Uncertainty in LLM-Based Systems","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09863","citing_title":"Nautilus Compass: Black-box Persona Drift Detection for Production LLM Agents","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":246,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QN3ABX2IIPHFIXAKK2GJAU6ZUP","json":"https://pith.science/pith/QN3ABX2IIPHFIXAKK2GJAU6ZUP.json","graph_json":"https://pith.science/api/pith-number/QN3ABX2IIPHFIXAKK2GJAU6ZUP/graph.json","events_json":"https://pith.science/api/pith-number/QN3ABX2IIPHFIXAKK2GJAU6ZUP/events.json","paper":"https://pith.science/paper/QN3ABX2I"},"agent_actions":{"view_html":"https://pith.science/pith/QN3ABX2IIPHFIXAKK2GJAU6ZUP","download_json":"https://pith.science/pith/QN3ABX2IIPHFIXAKK2GJAU6ZUP.json","view_paper":"https://pith.science/paper/QN3ABX2I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01257&json=true","fetch_graph":"https://pith.science/api/pith-number/QN3ABX2IIPHFIXAKK2GJAU6ZUP/graph.json","fetch_events":"https://pith.science/api/pith-number/QN3ABX2IIPHFIXAKK2GJAU6ZUP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QN3ABX2IIPHFIXAKK2GJAU6ZUP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QN3ABX2IIPHFIXAKK2GJAU6ZUP/action/storage_attestation","attest_author":"https://pith.science/pith/QN3ABX2IIPHFIXAKK2GJAU6ZUP/action/author_attestation","sign_citation":"https://pith.science/pith/QN3ABX2IIPHFIXAKK2GJAU6ZUP/action/citation_signature","submit_replication":"https://pith.science/pith/QN3ABX2IIPHFIXAKK2GJAU6ZUP/action/replication_record"}},"created_at":"2026-07-05T10:25:10.899806+00:00","updated_at":"2026-07-05T10:25:10.899806+00:00"}