{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D3MUCCGIW2A6WWHWFTSO5HH7WE","short_pith_number":"pith:D3MUCCGI","schema_version":"1.0","canonical_sha256":"1ed94108c8b681eb58f62ce4ee9cffb10d51d4eb6f1ef558c11c71c13cf7ca58","source":{"kind":"arxiv","id":"2405.21040","version":1},"attestation_state":"computed","paper":{"title":"Direct Alignment of Language Models via Quality-Aware Self-Refinement","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"James T. Kwok, Runsheng Yu, Xiaoqi Jiao, Yong Wang, Youzhi Zhang","submitted_at":"2024-05-31T17:31:18Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has been commonly used to align the behaviors of Large Language Models (LLMs) with human preferences. Recently, a popular alternative is Direct Policy Optimization (DPO), which replaces an LLM-based reward model with the policy itself, thus obviating the need for extra memory and training time to learn the reward model. However, DPO does not consider the relative qualities of the positive and negative responses, and can lead to sub-optimal training outcomes. To alleviate this problem, we investigate the use of intrinsic knowledge within the on-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.21040","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-31T17:31:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0a480228e3bbe1f16e6d6a92708f49434184fa11bf1ef02705c971d719f84c50","abstract_canon_sha256":"06bbc0ea1a5942d88507b131025d6ed8326e5e4c0f7d4f47a2fde8cf0dc6488a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:45.199345Z","signature_b64":"ljDwiSRacDQFOSI8gFDlciHBIikWy2AwrPwaatFKkZgjsRDjN3p293AoqRpBHYchUI0+hmrl+i8atwm8G03gAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1ed94108c8b681eb58f62ce4ee9cffb10d51d4eb6f1ef558c11c71c13cf7ca58","last_reissued_at":"2026-07-05T08:25:45.198888Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:45.198888Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Direct Alignment of Language Models via Quality-Aware Self-Refinement","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"James T. Kwok, Runsheng Yu, Xiaoqi Jiao, Yong Wang, Youzhi Zhang","submitted_at":"2024-05-31T17:31:18Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has been commonly used to align the behaviors of Large Language Models (LLMs) with human preferences. Recently, a popular alternative is Direct Policy Optimization (DPO), which replaces an LLM-based reward model with the policy itself, thus obviating the need for extra memory and training time to learn the reward model. However, DPO does not consider the relative qualities of the positive and negative responses, and can lead to sub-optimal training outcomes. To alleviate this problem, we investigate the use of intrinsic knowledge within the on-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.21040","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.21040/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.21040","created_at":"2026-07-05T08:25:45.198951+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.21040v1","created_at":"2026-07-05T08:25:45.198951+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.21040","created_at":"2026-07-05T08:25:45.198951+00:00"},{"alias_kind":"pith_short_12","alias_value":"D3MUCCGIW2A6","created_at":"2026-07-05T08:25:45.198951+00:00"},{"alias_kind":"pith_short_16","alias_value":"D3MUCCGIW2A6WWHW","created_at":"2026-07-05T08:25:45.198951+00:00"},{"alias_kind":"pith_short_8","alias_value":"D3MUCCGI","created_at":"2026-07-05T08:25:45.198951+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D3MUCCGIW2A6WWHWFTSO5HH7WE","json":"https://pith.science/pith/D3MUCCGIW2A6WWHWFTSO5HH7WE.json","graph_json":"https://pith.science/api/pith-number/D3MUCCGIW2A6WWHWFTSO5HH7WE/graph.json","events_json":"https://pith.science/api/pith-number/D3MUCCGIW2A6WWHWFTSO5HH7WE/events.json","paper":"https://pith.science/paper/D3MUCCGI"},"agent_actions":{"view_html":"https://pith.science/pith/D3MUCCGIW2A6WWHWFTSO5HH7WE","download_json":"https://pith.science/pith/D3MUCCGIW2A6WWHWFTSO5HH7WE.json","view_paper":"https://pith.science/paper/D3MUCCGI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.21040&json=true","fetch_graph":"https://pith.science/api/pith-number/D3MUCCGIW2A6WWHWFTSO5HH7WE/graph.json","fetch_events":"https://pith.science/api/pith-number/D3MUCCGIW2A6WWHWFTSO5HH7WE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D3MUCCGIW2A6WWHWFTSO5HH7WE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D3MUCCGIW2A6WWHWFTSO5HH7WE/action/storage_attestation","attest_author":"https://pith.science/pith/D3MUCCGIW2A6WWHWFTSO5HH7WE/action/author_attestation","sign_citation":"https://pith.science/pith/D3MUCCGIW2A6WWHWFTSO5HH7WE/action/citation_signature","submit_replication":"https://pith.science/pith/D3MUCCGIW2A6WWHWFTSO5HH7WE/action/replication_record"}},"created_at":"2026-07-05T08:25:45.198951+00:00","updated_at":"2026-07-05T08:25:45.198951+00:00"}