{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CTFQPF3IGN74OXJT5AKTFWUKH5","short_pith_number":"pith:CTFQPF3I","schema_version":"1.0","canonical_sha256":"14cb079768337fc75d33e81532da8a3f702655b09e9479da6d87edcf471ccc55","source":{"kind":"arxiv","id":"2402.02423","version":2},"attestation_state":"computed","paper":{"title":"Uni-RLHF: Universal Platform and Benchmark Suite for Reinforcement Learning with Diverse Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC","cs.RO"],"primary_cat":"cs.LG","authors_text":"Hebin Liang, Jianye Hao, Jinyi Liu, Kai Zhao, Yan Zheng, Yifu Yuan, Yi Ma, Zhixin Feng, Zibin Dong","submitted_at":"2024-02-04T09:40:22Z","abstract_excerpt":"Reinforcement Learning with Human Feedback (RLHF) has received significant attention for performing tasks without the need for costly manual reward design by aligning human preferences. It is crucial to consider diverse human feedback types and various learning methods in different environments. However, quantifying progress in RLHF with diverse feedback is challenging due to the lack of standardized annotation platforms and widely used unified benchmarks. To bridge this gap, we introduce Uni-RLHF, a comprehensive system implementation tailored for RLHF. It aims to provide a complete workflow "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.02423","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-04T09:40:22Z","cross_cats_sorted":["cs.AI","cs.HC","cs.RO"],"title_canon_sha256":"e12ea742123657d1b102224e3be2a883ef448291242cf8a12eea5274d687c9c3","abstract_canon_sha256":"cfbd955cec4ed064136638bc9b9f095c98606ff81c7f8c4f0937fc23a9fa4cfe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:00:01.066528Z","signature_b64":"+cxsdq+vk6rnJP+UouuGwDxWZmou3ygY3nFvG5Bx+7FbQvji4WXm3m/u99xW3RvhudLUx/7U7mYlyw4ltppzAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"14cb079768337fc75d33e81532da8a3f702655b09e9479da6d87edcf471ccc55","last_reissued_at":"2026-07-05T08:00:01.065952Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:00:01.065952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Uni-RLHF: Universal Platform and Benchmark Suite for Reinforcement Learning with Diverse Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC","cs.RO"],"primary_cat":"cs.LG","authors_text":"Hebin Liang, Jianye Hao, Jinyi Liu, Kai Zhao, Yan Zheng, Yifu Yuan, Yi Ma, Zhixin Feng, Zibin Dong","submitted_at":"2024-02-04T09:40:22Z","abstract_excerpt":"Reinforcement Learning with Human Feedback (RLHF) has received significant attention for performing tasks without the need for costly manual reward design by aligning human preferences. It is crucial to consider diverse human feedback types and various learning methods in different environments. However, quantifying progress in RLHF with diverse feedback is challenging due to the lack of standardized annotation platforms and widely used unified benchmarks. To bridge this gap, we introduce Uni-RLHF, a comprehensive system implementation tailored for RLHF. It aims to provide a complete workflow "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.02423","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.02423/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.02423","created_at":"2026-07-05T08:00:01.066014+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.02423v2","created_at":"2026-07-05T08:00:01.066014+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.02423","created_at":"2026-07-05T08:00:01.066014+00:00"},{"alias_kind":"pith_short_12","alias_value":"CTFQPF3IGN74","created_at":"2026-07-05T08:00:01.066014+00:00"},{"alias_kind":"pith_short_16","alias_value":"CTFQPF3IGN74OXJT","created_at":"2026-07-05T08:00:01.066014+00:00"},{"alias_kind":"pith_short_8","alias_value":"CTFQPF3I","created_at":"2026-07-05T08:00:01.066014+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.20782","citing_title":"Integrating Human Feedback into a Reinforcement Learning-Based Framework for Adaptive User Interfaces","ref_index":38,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CTFQPF3IGN74OXJT5AKTFWUKH5","json":"https://pith.science/pith/CTFQPF3IGN74OXJT5AKTFWUKH5.json","graph_json":"https://pith.science/api/pith-number/CTFQPF3IGN74OXJT5AKTFWUKH5/graph.json","events_json":"https://pith.science/api/pith-number/CTFQPF3IGN74OXJT5AKTFWUKH5/events.json","paper":"https://pith.science/paper/CTFQPF3I"},"agent_actions":{"view_html":"https://pith.science/pith/CTFQPF3IGN74OXJT5AKTFWUKH5","download_json":"https://pith.science/pith/CTFQPF3IGN74OXJT5AKTFWUKH5.json","view_paper":"https://pith.science/paper/CTFQPF3I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.02423&json=true","fetch_graph":"https://pith.science/api/pith-number/CTFQPF3IGN74OXJT5AKTFWUKH5/graph.json","fetch_events":"https://pith.science/api/pith-number/CTFQPF3IGN74OXJT5AKTFWUKH5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CTFQPF3IGN74OXJT5AKTFWUKH5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CTFQPF3IGN74OXJT5AKTFWUKH5/action/storage_attestation","attest_author":"https://pith.science/pith/CTFQPF3IGN74OXJT5AKTFWUKH5/action/author_attestation","sign_citation":"https://pith.science/pith/CTFQPF3IGN74OXJT5AKTFWUKH5/action/citation_signature","submit_replication":"https://pith.science/pith/CTFQPF3IGN74OXJT5AKTFWUKH5/action/replication_record"}},"created_at":"2026-07-05T08:00:01.066014+00:00","updated_at":"2026-07-05T08:00:01.066014+00:00"}