{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5SYIOFBQWUBP6AGWAFGMDXPKW3","short_pith_number":"pith:5SYIOFBQ","schema_version":"1.0","canonical_sha256":"ecb0871430b502ff00d6014cc1ddeab6c9de3d2993cf1597fa03130df702c429","source":{"kind":"arxiv","id":"2412.05469","version":1},"attestation_state":"computed","paper":{"title":"Multi-Objective Alignment of Large Language Models Through Hypervolume Maximization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aniket Deshmukh, Anusha Lalitha, Branislav Kveton, Sailik Sengupta, Subhojyoti Mukherjee","submitted_at":"2024-12-06T23:51:47Z","abstract_excerpt":"Multi-objective alignment from human feedback (MOAHF) in large language models (LLMs) is a challenging problem as human preferences are complex, multifaceted, and often conflicting. Recent works on MOAHF considered a-priori multi-objective optimization (MOO), where human preferences are known at training or inference time. In contrast, when human preferences are unknown or difficult to quantify, a natural approach is to cover the Pareto front by multiple diverse solutions. We propose an algorithm HaM for learning diverse LLM policies that maximizes their hypervolume. This is the first applicat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.05469","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-06T23:51:47Z","cross_cats_sorted":[],"title_canon_sha256":"560a140201b8158d968a3dd8c7ccc0c0e1b7bebc5b5a71e077b0e05fee331b02","abstract_canon_sha256":"33f4742a46c71501b8387c7aade6fb4845c1859075f439d7c686c1fcc62205ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:46:03.776216Z","signature_b64":"vvouB2FBu3dkKL+zbY14Aw2ha7wYGh2YkcaDJrv7gkMxLU4MZKf1nCyAUxIKQx6MehjkzhSYeVzg0VMOCajODQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ecb0871430b502ff00d6014cc1ddeab6c9de3d2993cf1597fa03130df702c429","last_reissued_at":"2026-07-05T09:46:03.775764Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:46:03.775764Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Objective Alignment of Large Language Models Through Hypervolume Maximization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aniket Deshmukh, Anusha Lalitha, Branislav Kveton, Sailik Sengupta, Subhojyoti Mukherjee","submitted_at":"2024-12-06T23:51:47Z","abstract_excerpt":"Multi-objective alignment from human feedback (MOAHF) in large language models (LLMs) is a challenging problem as human preferences are complex, multifaceted, and often conflicting. Recent works on MOAHF considered a-priori multi-objective optimization (MOO), where human preferences are known at training or inference time. In contrast, when human preferences are unknown or difficult to quantify, a natural approach is to cover the Pareto front by multiple diverse solutions. We propose an algorithm HaM for learning diverse LLM policies that maximizes their hypervolume. This is the first applicat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.05469","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.05469/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.05469","created_at":"2026-07-05T09:46:03.775827+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.05469v1","created_at":"2026-07-05T09:46:03.775827+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.05469","created_at":"2026-07-05T09:46:03.775827+00:00"},{"alias_kind":"pith_short_12","alias_value":"5SYIOFBQWUBP","created_at":"2026-07-05T09:46:03.775827+00:00"},{"alias_kind":"pith_short_16","alias_value":"5SYIOFBQWUBP6AGW","created_at":"2026-07-05T09:46:03.775827+00:00"},{"alias_kind":"pith_short_8","alias_value":"5SYIOFBQ","created_at":"2026-07-05T09:46:03.775827+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07988","citing_title":"PAFO: Pareto Fairness Optimization for Personalized Reward Modeling","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26579","citing_title":"Focal Reward: Balanced Reinforcement Learning under Rubric-Based Rewards","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19330","citing_title":"MOCHA: Multi-Objective Chebyshev Annealing for Agent Skill Optimization","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5SYIOFBQWUBP6AGWAFGMDXPKW3","json":"https://pith.science/pith/5SYIOFBQWUBP6AGWAFGMDXPKW3.json","graph_json":"https://pith.science/api/pith-number/5SYIOFBQWUBP6AGWAFGMDXPKW3/graph.json","events_json":"https://pith.science/api/pith-number/5SYIOFBQWUBP6AGWAFGMDXPKW3/events.json","paper":"https://pith.science/paper/5SYIOFBQ"},"agent_actions":{"view_html":"https://pith.science/pith/5SYIOFBQWUBP6AGWAFGMDXPKW3","download_json":"https://pith.science/pith/5SYIOFBQWUBP6AGWAFGMDXPKW3.json","view_paper":"https://pith.science/paper/5SYIOFBQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.05469&json=true","fetch_graph":"https://pith.science/api/pith-number/5SYIOFBQWUBP6AGWAFGMDXPKW3/graph.json","fetch_events":"https://pith.science/api/pith-number/5SYIOFBQWUBP6AGWAFGMDXPKW3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5SYIOFBQWUBP6AGWAFGMDXPKW3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5SYIOFBQWUBP6AGWAFGMDXPKW3/action/storage_attestation","attest_author":"https://pith.science/pith/5SYIOFBQWUBP6AGWAFGMDXPKW3/action/author_attestation","sign_citation":"https://pith.science/pith/5SYIOFBQWUBP6AGWAFGMDXPKW3/action/citation_signature","submit_replication":"https://pith.science/pith/5SYIOFBQWUBP6AGWAFGMDXPKW3/action/replication_record"}},"created_at":"2026-07-05T09:46:03.775827+00:00","updated_at":"2026-07-05T09:46:03.775827+00:00"}