{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LVKTAOYDYYBGIPVSBHABYFYF54","short_pith_number":"pith:LVKTAOYD","schema_version":"1.0","canonical_sha256":"5d55303b03c602643eb209c01c1705ef37cb347ee8b56b65d5e31da8c1350fcb","source":{"kind":"arxiv","id":"2310.11523","version":2},"attestation_state":"computed","paper":{"title":"Group Preference Optimization: Few-Shot Alignment of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aditya Grover, John Dang, Siyan Zhao","submitted_at":"2023-10-17T18:41:57Z","abstract_excerpt":"Many applications of large language models (LLMs), ranging from chatbots to creative writing, require nuanced subjective judgments that can differ significantly across different groups. Existing alignment algorithms can be expensive to align for each group, requiring prohibitive amounts of group-specific preference data and computation for real-world use cases. We introduce Group Preference Optimization (GPO), an alignment framework that steers language models to preferences of individual groups in a few-shot manner. In GPO, we augment the base LLM with an independent transformer module traine"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.11523","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-17T18:41:57Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a0cc88434bcf2c66430433d0bba11ded9934f49d9d768ae7221e4c4a79595039","abstract_canon_sha256":"643b8c4fda20e6ec77a379966b49ec66043e5c8d6ed93c51630cbf8c166e9de9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:22.001648Z","signature_b64":"1JkQ07fJboc2IaC6lVhwEEyMXxEn/yELvXXVnCMHFvQBzcm7jtvQueeIKAfwHnwZPfgGXp2CoTZ3BrHia++yBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d55303b03c602643eb209c01c1705ef37cb347ee8b56b65d5e31da8c1350fcb","last_reissued_at":"2026-07-05T09:20:22.001113Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:22.001113Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Group Preference Optimization: Few-Shot Alignment of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aditya Grover, John Dang, Siyan Zhao","submitted_at":"2023-10-17T18:41:57Z","abstract_excerpt":"Many applications of large language models (LLMs), ranging from chatbots to creative writing, require nuanced subjective judgments that can differ significantly across different groups. Existing alignment algorithms can be expensive to align for each group, requiring prohibitive amounts of group-specific preference data and computation for real-world use cases. We introduce Group Preference Optimization (GPO), an alignment framework that steers language models to preferences of individual groups in a few-shot manner. In GPO, we augment the base LLM with an independent transformer module traine"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.11523","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.11523/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.11523","created_at":"2026-07-05T09:20:22.001171+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.11523v2","created_at":"2026-07-05T09:20:22.001171+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.11523","created_at":"2026-07-05T09:20:22.001171+00:00"},{"alias_kind":"pith_short_12","alias_value":"LVKTAOYDYYBG","created_at":"2026-07-05T09:20:22.001171+00:00"},{"alias_kind":"pith_short_16","alias_value":"LVKTAOYDYYBGIPVS","created_at":"2026-07-05T09:20:22.001171+00:00"},{"alias_kind":"pith_short_8","alias_value":"LVKTAOYD","created_at":"2026-07-05T09:20:22.001171+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11583","citing_title":"Beyond the Golden Teacher: Enhancing Graph Learning through LLM-GNN Co-teaching","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07988","citing_title":"PAFO: Pareto Fairness Optimization for Personalized Reward Modeling","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17881","citing_title":"POPI: Personalizing LLMs via Optimized Natural Language Preference Inference","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26202","citing_title":"What's In My Human Feedback? Learning Interpretable Descriptions of Preference Data","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2402.05070","citing_title":"A Roadmap to Pluralistic Alignment","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25895","citing_title":"Three Models of RLHF Annotation: Extension, Evidence, and Authority","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09876","citing_title":"Efficient Personalization of Generative User Interfaces","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07343","citing_title":"Personalized RewardBench: Evaluating Reward Models with Human Aligned Personalization","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LVKTAOYDYYBGIPVSBHABYFYF54","json":"https://pith.science/pith/LVKTAOYDYYBGIPVSBHABYFYF54.json","graph_json":"https://pith.science/api/pith-number/LVKTAOYDYYBGIPVSBHABYFYF54/graph.json","events_json":"https://pith.science/api/pith-number/LVKTAOYDYYBGIPVSBHABYFYF54/events.json","paper":"https://pith.science/paper/LVKTAOYD"},"agent_actions":{"view_html":"https://pith.science/pith/LVKTAOYDYYBGIPVSBHABYFYF54","download_json":"https://pith.science/pith/LVKTAOYDYYBGIPVSBHABYFYF54.json","view_paper":"https://pith.science/paper/LVKTAOYD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.11523&json=true","fetch_graph":"https://pith.science/api/pith-number/LVKTAOYDYYBGIPVSBHABYFYF54/graph.json","fetch_events":"https://pith.science/api/pith-number/LVKTAOYDYYBGIPVSBHABYFYF54/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LVKTAOYDYYBGIPVSBHABYFYF54/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LVKTAOYDYYBGIPVSBHABYFYF54/action/storage_attestation","attest_author":"https://pith.science/pith/LVKTAOYDYYBGIPVSBHABYFYF54/action/author_attestation","sign_citation":"https://pith.science/pith/LVKTAOYDYYBGIPVSBHABYFYF54/action/citation_signature","submit_replication":"https://pith.science/pith/LVKTAOYDYYBGIPVSBHABYFYF54/action/replication_record"}},"created_at":"2026-07-05T09:20:22.001171+00:00","updated_at":"2026-07-05T09:20:22.001171+00:00"}