{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RAB2QRFBTOWSOXCOEZ3ZRAIS73","short_pith_number":"pith:RAB2QRFB","schema_version":"1.0","canonical_sha256":"8803a844a19bad275c4e2677988112feea115a15f9c95fb32e623d3732b930c8","source":{"kind":"arxiv","id":"2505.20359","version":2},"attestation_state":"computed","paper":{"title":"Risk-aware Direct Preference Optimization under Nested Risk Measure","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Huizhong Song, Jun Wang, Lijun Zhang, Lin Li, Wei Wei, Yajie Qi, Yaodong Yang","submitted_at":"2025-05-26T08:01:37Z","abstract_excerpt":"When fine-tuning pre-trained Large Language Models (LLMs) to align with human values and intentions, maximizing the estimated reward can lead to superior performance, but it also introduces potential risks due to deviations from the reference model's intended behavior. Most existing methods typically introduce KL divergence to constrain deviations between the trained model and the reference model; however, this may not be sufficient in certain applications that require tight risk control. In this paper, we introduce Risk-aware Direct Preference Optimization (Ra-DPO), a novel approach that inco"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.20359","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-26T08:01:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"03afc959742db52b81bbfbc8d3230ab191741f76d0dcee2152cbf5260a0efed3","abstract_canon_sha256":"003f8c6e257383beff5c1b38e81eeb4ce154ef90833a98985d7d43abef00f521"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:46.981677Z","signature_b64":"hReSZZy/+lRb4lfD9gPDVK4hdNAXqk/8nfmCDjjuy3Bo9z0V802JfkNil68QLbXW3EoyuCkPbyYZQcuY3hyBDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8803a844a19bad275c4e2677988112feea115a15f9c95fb32e623d3732b930c8","last_reissued_at":"2026-07-05T11:11:46.981142Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:46.981142Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Risk-aware Direct Preference Optimization under Nested Risk Measure","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Huizhong Song, Jun Wang, Lijun Zhang, Lin Li, Wei Wei, Yajie Qi, Yaodong Yang","submitted_at":"2025-05-26T08:01:37Z","abstract_excerpt":"When fine-tuning pre-trained Large Language Models (LLMs) to align with human values and intentions, maximizing the estimated reward can lead to superior performance, but it also introduces potential risks due to deviations from the reference model's intended behavior. Most existing methods typically introduce KL divergence to constrain deviations between the trained model and the reference model; however, this may not be sufficient in certain applications that require tight risk control. In this paper, we introduce Risk-aware Direct Preference Optimization (Ra-DPO), a novel approach that inco"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.20359","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.20359/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.20359","created_at":"2026-07-05T11:11:46.981212+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.20359v2","created_at":"2026-07-05T11:11:46.981212+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.20359","created_at":"2026-07-05T11:11:46.981212+00:00"},{"alias_kind":"pith_short_12","alias_value":"RAB2QRFBTOWS","created_at":"2026-07-05T11:11:46.981212+00:00"},{"alias_kind":"pith_short_16","alias_value":"RAB2QRFBTOWSOXCO","created_at":"2026-07-05T11:11:46.981212+00:00"},{"alias_kind":"pith_short_8","alias_value":"RAB2QRFB","created_at":"2026-07-05T11:11:46.981212+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RAB2QRFBTOWSOXCOEZ3ZRAIS73","json":"https://pith.science/pith/RAB2QRFBTOWSOXCOEZ3ZRAIS73.json","graph_json":"https://pith.science/api/pith-number/RAB2QRFBTOWSOXCOEZ3ZRAIS73/graph.json","events_json":"https://pith.science/api/pith-number/RAB2QRFBTOWSOXCOEZ3ZRAIS73/events.json","paper":"https://pith.science/paper/RAB2QRFB"},"agent_actions":{"view_html":"https://pith.science/pith/RAB2QRFBTOWSOXCOEZ3ZRAIS73","download_json":"https://pith.science/pith/RAB2QRFBTOWSOXCOEZ3ZRAIS73.json","view_paper":"https://pith.science/paper/RAB2QRFB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.20359&json=true","fetch_graph":"https://pith.science/api/pith-number/RAB2QRFBTOWSOXCOEZ3ZRAIS73/graph.json","fetch_events":"https://pith.science/api/pith-number/RAB2QRFBTOWSOXCOEZ3ZRAIS73/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RAB2QRFBTOWSOXCOEZ3ZRAIS73/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RAB2QRFBTOWSOXCOEZ3ZRAIS73/action/storage_attestation","attest_author":"https://pith.science/pith/RAB2QRFBTOWSOXCOEZ3ZRAIS73/action/author_attestation","sign_citation":"https://pith.science/pith/RAB2QRFBTOWSOXCOEZ3ZRAIS73/action/citation_signature","submit_replication":"https://pith.science/pith/RAB2QRFBTOWSOXCOEZ3ZRAIS73/action/replication_record"}},"created_at":"2026-07-05T11:11:46.981212+00:00","updated_at":"2026-07-05T11:11:46.981212+00:00"}