{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:7OVGIHWBXMRKDBQSGZ45TY4UTK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"915a76c12ae7cd4eabcb0d053dc1adc1738e413055990b7f953a2e759b9a06a8","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-30T17:50:04Z","title_canon_sha256":"9fda535b3d0de6e483609dd46d3b214b98b8620a8d70ae8953b6b0ea93df2f8a"},"schema_version":"1.0","source":{"id":"2405.20304","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.20304","created_at":"2026-07-05T08:25:18Z"},{"alias_kind":"arxiv_version","alias_value":"2405.20304v1","created_at":"2026-07-05T08:25:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20304","created_at":"2026-07-05T08:25:18Z"},{"alias_kind":"pith_short_12","alias_value":"7OVGIHWBXMRK","created_at":"2026-07-05T08:25:18Z"},{"alias_kind":"pith_short_16","alias_value":"7OVGIHWBXMRKDBQS","created_at":"2026-07-05T08:25:18Z"},{"alias_kind":"pith_short_8","alias_value":"7OVGIHWB","created_at":"2026-07-05T08:25:18Z"}],"graph_snapshots":[{"event_id":"sha256:f1a52c3e43ef77258efa062e02a1ebd570fb52c4060854de0f36babf76b1c08a","target":"graph","created_at":"2026-07-05T08:25:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.20304/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Adapting large language models (LLMs) for specific tasks usually involves fine-tuning through reinforcement learning with human feedback (RLHF) on preference data. While these data often come from diverse labelers' groups (e.g., different demographics, ethnicities, company teams, etc.), traditional RLHF approaches adopt a \"one-size-fits-all\" approach, i.e., they indiscriminately assume and optimize a single preference model, thus not being robust to unique characteristics and needs of the various groups. To address this limitation, we propose a novel Group Robust Preference Optimization (GRPO)","authors_text":"Haitham Bou Ammar, Iason Chaimalas, Ilija Bogunovic, Pier Giuseppe Sessa, Shyam Sundhar Ramesh, Viraj Mehta, Yifan Hu","cross_cats":["cs.LG"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-30T17:50:04Z","title":"Group Robust Preference Optimization in Reward-free RLHF"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20304","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:8fc3e11c63af60a4f2055905ec2d557996435476db0786e82c98e2a826eeccc3","target":"record","created_at":"2026-07-05T08:25:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"915a76c12ae7cd4eabcb0d053dc1adc1738e413055990b7f953a2e759b9a06a8","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-30T17:50:04Z","title_canon_sha256":"9fda535b3d0de6e483609dd46d3b214b98b8620a8d70ae8953b6b0ea93df2f8a"},"schema_version":"1.0","source":{"id":"2405.20304","kind":"arxiv","version":1}},"canonical_sha256":"fbaa641ec1bb22a186123679d9e3949a8dc14a96c5d0e9f7280c452a85c0cc4b","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"fbaa641ec1bb22a186123679d9e3949a8dc14a96c5d0e9f7280c452a85c0cc4b","first_computed_at":"2026-07-05T08:25:18.603125Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:25:18.603125Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"V3TP83bbePYl9O9hwTSOSdD3PMGIUM7iXFZw14W3kMVJvShiwWWLCQGiFa9iuKUZXVUGD8yWLnL/ckswnJV2CA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:25:18.603542Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.20304","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:8fc3e11c63af60a4f2055905ec2d557996435476db0786e82c98e2a826eeccc3","sha256:f1a52c3e43ef77258efa062e02a1ebd570fb52c4060854de0f36babf76b1c08a"],"state_sha256":"55c84db226d4f65e980641a026c71b5eca92e264a69ef4a844d92d6501e1d520"}