{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6JSQO7CAKIQHSGWZUZ7SXG4ZOB","short_pith_number":"pith:6JSQO7CA","schema_version":"1.0","canonical_sha256":"f265077c405220791ad9a67f2b9b9970442b713823e123b8afe236ac8d22e3ab","source":{"kind":"arxiv","id":"2503.21819","version":1},"attestation_state":"computed","paper":{"title":"Optimizing Safe and Aligned Language Generation: A Multi-Objective GRPO Approach","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Victor Bian, Xuying Li, Yuji Kosuga, Zhuo Li","submitted_at":"2025-03-26T05:50:33Z","abstract_excerpt":"Aligning large language models (LLMs) with human values and safety constraints is challenging, especially when objectives like helpfulness, truthfulness, and avoidance of harm conflict. Reinforcement Learning from Human Feedback (RLHF) has achieved notable success in steering models, but is complex and can be unstable. Recent approaches such as Direct Preference Optimization (DPO) simplify preference-based fine-tuning but may introduce bias or trade-off certain objectives~\\cite{dpo}. In this work, we propose a Group Relative Policy Optimization (GRPO) framework with a multi-label reward regres"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.21819","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-26T05:50:33Z","cross_cats_sorted":[],"title_canon_sha256":"ac2120a24f5a5aa04ef818f2251df8564f60a1f97c6807b2eb09899bd08db06b","abstract_canon_sha256":"c5f191ad8777408b4aa6ac398e4757d0f7783e2d7571eadadbbff74911e9922b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:42.583496Z","signature_b64":"QZT22XUct25VIEz/zC4BvZBNJBA8Nlmx9OHg7WIxDMy2kouA6KAgbgY4WY+3ypNTBx1LLWr732r0rYuMnyTCDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f265077c405220791ad9a67f2b9b9970442b713823e123b8afe236ac8d22e3ab","last_reissued_at":"2026-07-05T10:40:42.583030Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:42.583030Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimizing Safe and Aligned Language Generation: A Multi-Objective GRPO Approach","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Victor Bian, Xuying Li, Yuji Kosuga, Zhuo Li","submitted_at":"2025-03-26T05:50:33Z","abstract_excerpt":"Aligning large language models (LLMs) with human values and safety constraints is challenging, especially when objectives like helpfulness, truthfulness, and avoidance of harm conflict. Reinforcement Learning from Human Feedback (RLHF) has achieved notable success in steering models, but is complex and can be unstable. Recent approaches such as Direct Preference Optimization (DPO) simplify preference-based fine-tuning but may introduce bias or trade-off certain objectives~\\cite{dpo}. In this work, we propose a Group Relative Policy Optimization (GRPO) framework with a multi-label reward regres"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.21819","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.21819/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.21819","created_at":"2026-07-05T10:40:42.583084+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.21819v1","created_at":"2026-07-05T10:40:42.583084+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.21819","created_at":"2026-07-05T10:40:42.583084+00:00"},{"alias_kind":"pith_short_12","alias_value":"6JSQO7CAKIQH","created_at":"2026-07-05T10:40:42.583084+00:00"},{"alias_kind":"pith_short_16","alias_value":"6JSQO7CAKIQHSGWZ","created_at":"2026-07-05T10:40:42.583084+00:00"},{"alias_kind":"pith_short_8","alias_value":"6JSQO7CA","created_at":"2026-07-05T10:40:42.583084+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24005","citing_title":"LC-ERD: Mining Latent Logic for Self-Evolving Reasoning via Consistency-Regulated Reward Decomposition","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05750","citing_title":"RVPO: Risk-Sensitive Alignment via Variance Regularization","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6JSQO7CAKIQHSGWZUZ7SXG4ZOB","json":"https://pith.science/pith/6JSQO7CAKIQHSGWZUZ7SXG4ZOB.json","graph_json":"https://pith.science/api/pith-number/6JSQO7CAKIQHSGWZUZ7SXG4ZOB/graph.json","events_json":"https://pith.science/api/pith-number/6JSQO7CAKIQHSGWZUZ7SXG4ZOB/events.json","paper":"https://pith.science/paper/6JSQO7CA"},"agent_actions":{"view_html":"https://pith.science/pith/6JSQO7CAKIQHSGWZUZ7SXG4ZOB","download_json":"https://pith.science/pith/6JSQO7CAKIQHSGWZUZ7SXG4ZOB.json","view_paper":"https://pith.science/paper/6JSQO7CA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.21819&json=true","fetch_graph":"https://pith.science/api/pith-number/6JSQO7CAKIQHSGWZUZ7SXG4ZOB/graph.json","fetch_events":"https://pith.science/api/pith-number/6JSQO7CAKIQHSGWZUZ7SXG4ZOB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6JSQO7CAKIQHSGWZUZ7SXG4ZOB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6JSQO7CAKIQHSGWZUZ7SXG4ZOB/action/storage_attestation","attest_author":"https://pith.science/pith/6JSQO7CAKIQHSGWZUZ7SXG4ZOB/action/author_attestation","sign_citation":"https://pith.science/pith/6JSQO7CAKIQHSGWZUZ7SXG4ZOB/action/citation_signature","submit_replication":"https://pith.science/pith/6JSQO7CAKIQHSGWZUZ7SXG4ZOB/action/replication_record"}},"created_at":"2026-07-05T10:40:42.583084+00:00","updated_at":"2026-07-05T10:40:42.583084+00:00"}