{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XQZ5G2ZUOSX4A3DHYXEHDEN5IW","short_pith_number":"pith:XQZ5G2ZU","schema_version":"1.0","canonical_sha256":"bc33d36b3474afc06c67c5c87191bd45a49e6a79cae448590e6e4e6e30e45047","source":{"kind":"arxiv","id":"2409.06691","version":3},"attestation_state":"computed","paper":{"title":"Geometric-Averaged Preference Optimization for Soft Preference Labels","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aleksandra Faust, Heiga Zen, Hiroki Furuta, Izzeddin Gur, Kuang-Huei Lee, Shixiang Shane Gu, Yutaka Matsuo","submitted_at":"2024-09-10T17:54:28Z","abstract_excerpt":"Many algorithms for aligning LLMs with human preferences assume that human preferences are binary and deterministic. However, human preferences can vary across individuals, and therefore should be represented distributionally. In this work, we introduce the distributional soft preference labels and improve Direct Preference Optimization (DPO) with a weighted geometric average of the LLM output likelihood in the loss function. This approach adjusts the scale of learning loss based on the soft labels such that the loss would approach zero when the responses are closer to equally preferred. This "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.06691","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-10T17:54:28Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"ce52f66f435f0fa3e7ab26dd06b52d8963434b59ad7dcec63bb6d89b2527bf8e","abstract_canon_sha256":"5a211d278e9bb6ca2bd69d079903c376c23dc337a03a663713f60b9814b725ee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:12.076194Z","signature_b64":"NP7/vSBKrDviRsZ1QDHTVI7Y8HqGh9Zvipde55QH3st2n8Vn6NbX9Jd9kLXq9YMFGZPO3NObhk9bPs1Dh44RBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bc33d36b3474afc06c67c5c87191bd45a49e6a79cae448590e6e4e6e30e45047","last_reissued_at":"2026-07-05T09:55:12.075760Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:12.075760Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Geometric-Averaged Preference Optimization for Soft Preference Labels","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aleksandra Faust, Heiga Zen, Hiroki Furuta, Izzeddin Gur, Kuang-Huei Lee, Shixiang Shane Gu, Yutaka Matsuo","submitted_at":"2024-09-10T17:54:28Z","abstract_excerpt":"Many algorithms for aligning LLMs with human preferences assume that human preferences are binary and deterministic. However, human preferences can vary across individuals, and therefore should be represented distributionally. In this work, we introduce the distributional soft preference labels and improve Direct Preference Optimization (DPO) with a weighted geometric average of the LLM output likelihood in the loss function. This approach adjusts the scale of learning loss based on the soft labels such that the loss would approach zero when the responses are closer to equally preferred. This "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.06691","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.06691/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.06691","created_at":"2026-07-05T09:55:12.075823+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.06691v3","created_at":"2026-07-05T09:55:12.075823+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.06691","created_at":"2026-07-05T09:55:12.075823+00:00"},{"alias_kind":"pith_short_12","alias_value":"XQZ5G2ZUOSX4","created_at":"2026-07-05T09:55:12.075823+00:00"},{"alias_kind":"pith_short_16","alias_value":"XQZ5G2ZUOSX4A3DH","created_at":"2026-07-05T09:55:12.075823+00:00"},{"alias_kind":"pith_short_8","alias_value":"XQZ5G2ZU","created_at":"2026-07-05T09:55:12.075823+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.09572","citing_title":"Plan-and-Act: Improving Planning of Agents for Long-Horizon Tasks","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XQZ5G2ZUOSX4A3DHYXEHDEN5IW","json":"https://pith.science/pith/XQZ5G2ZUOSX4A3DHYXEHDEN5IW.json","graph_json":"https://pith.science/api/pith-number/XQZ5G2ZUOSX4A3DHYXEHDEN5IW/graph.json","events_json":"https://pith.science/api/pith-number/XQZ5G2ZUOSX4A3DHYXEHDEN5IW/events.json","paper":"https://pith.science/paper/XQZ5G2ZU"},"agent_actions":{"view_html":"https://pith.science/pith/XQZ5G2ZUOSX4A3DHYXEHDEN5IW","download_json":"https://pith.science/pith/XQZ5G2ZUOSX4A3DHYXEHDEN5IW.json","view_paper":"https://pith.science/paper/XQZ5G2ZU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.06691&json=true","fetch_graph":"https://pith.science/api/pith-number/XQZ5G2ZUOSX4A3DHYXEHDEN5IW/graph.json","fetch_events":"https://pith.science/api/pith-number/XQZ5G2ZUOSX4A3DHYXEHDEN5IW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XQZ5G2ZUOSX4A3DHYXEHDEN5IW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XQZ5G2ZUOSX4A3DHYXEHDEN5IW/action/storage_attestation","attest_author":"https://pith.science/pith/XQZ5G2ZUOSX4A3DHYXEHDEN5IW/action/author_attestation","sign_citation":"https://pith.science/pith/XQZ5G2ZUOSX4A3DHYXEHDEN5IW/action/citation_signature","submit_replication":"https://pith.science/pith/XQZ5G2ZUOSX4A3DHYXEHDEN5IW/action/replication_record"}},"created_at":"2026-07-05T09:55:12.075823+00:00","updated_at":"2026-07-05T09:55:12.075823+00:00"}