{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:73527J7JDIVSKROCF3WJRJDS4V","short_pith_number":"pith:73527J7J","schema_version":"1.0","canonical_sha256":"fefbafa7e91a2b2545c22eec98a472e55e0de1ca6c1dabce4148df227cc093ea","source":{"kind":"arxiv","id":"2407.15762","version":2},"attestation_state":"computed","paper":{"title":"Conditional Language Policy: A General Framework for Steerable Multi-Objective Finetuning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alekh Agarwal, Alexandre Ram\\'e, Amr Ahmed, Andrea Michi, Aranyak Mehta, Avinava Dubey, Christoph Dann, Edouard Leurent, Geoffrey Cideron, Hongkun Yu, Johan Ferret, Kaiwen Wang, Le Hou, L\\'eonard Hussenot, Marco Gelmi, Olivier Bachem, Raghav Gupta, Rahul Kidambi, Ryan Sullivan, Yunxuan Li","submitted_at":"2024-07-22T16:13:38Z","abstract_excerpt":"Reward-based finetuning is crucial for aligning language policies with intended behaviors (e.g., creativity and safety). A key challenge is to develop steerable language models that trade-off multiple (conflicting) objectives in a flexible and efficient manner. This paper presents Conditional Language Policy (CLP), a general framework for finetuning language models on multiple objectives. Building on techniques from multi-task training and parameter-efficient finetuning, CLP learn steerable models that effectively trade-off conflicting objectives at inference time. Notably, this does not requi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.15762","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-22T16:13:38Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"51b78c4139add8f20b630062e7332741a6bdef1bebd768349f4d37696921a324","abstract_canon_sha256":"2a0759efaca8da8f66d5a98f55d87558e808580a399b2e4125d2cfd8fd8065be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:37.050863Z","signature_b64":"vK/1iy4cGCfF5b3vO6O/Z6YhXRjvvWQo8aaua0Mhl7czTAUicYX/iTYFMFRMtrfF/7oQRXxDTMB6EopjjfAZCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fefbafa7e91a2b2545c22eec98a472e55e0de1ca6c1dabce4148df227cc093ea","last_reissued_at":"2026-07-05T09:24:37.050352Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:37.050352Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Conditional Language Policy: A General Framework for Steerable Multi-Objective Finetuning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alekh Agarwal, Alexandre Ram\\'e, Amr Ahmed, Andrea Michi, Aranyak Mehta, Avinava Dubey, Christoph Dann, Edouard Leurent, Geoffrey Cideron, Hongkun Yu, Johan Ferret, Kaiwen Wang, Le Hou, L\\'eonard Hussenot, Marco Gelmi, Olivier Bachem, Raghav Gupta, Rahul Kidambi, Ryan Sullivan, Yunxuan Li","submitted_at":"2024-07-22T16:13:38Z","abstract_excerpt":"Reward-based finetuning is crucial for aligning language policies with intended behaviors (e.g., creativity and safety). A key challenge is to develop steerable language models that trade-off multiple (conflicting) objectives in a flexible and efficient manner. This paper presents Conditional Language Policy (CLP), a general framework for finetuning language models on multiple objectives. Building on techniques from multi-task training and parameter-efficient finetuning, CLP learn steerable models that effectively trade-off conflicting objectives at inference time. Notably, this does not requi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.15762","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.15762/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.15762","created_at":"2026-07-05T09:24:37.050412+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.15762v2","created_at":"2026-07-05T09:24:37.050412+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.15762","created_at":"2026-07-05T09:24:37.050412+00:00"},{"alias_kind":"pith_short_12","alias_value":"73527J7JDIVS","created_at":"2026-07-05T09:24:37.050412+00:00"},{"alias_kind":"pith_short_16","alias_value":"73527J7JDIVSKROC","created_at":"2026-07-05T09:24:37.050412+00:00"},{"alias_kind":"pith_short_8","alias_value":"73527J7J","created_at":"2026-07-05T09:24:37.050412+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2408.07666","citing_title":"Model Merging in LLMs, MLLMs, and Beyond: Methods, Theories, Applications and Opportunities","ref_index":241,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11679","citing_title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11679","citing_title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","ref_index":67,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/73527J7JDIVSKROCF3WJRJDS4V","json":"https://pith.science/pith/73527J7JDIVSKROCF3WJRJDS4V.json","graph_json":"https://pith.science/api/pith-number/73527J7JDIVSKROCF3WJRJDS4V/graph.json","events_json":"https://pith.science/api/pith-number/73527J7JDIVSKROCF3WJRJDS4V/events.json","paper":"https://pith.science/paper/73527J7J"},"agent_actions":{"view_html":"https://pith.science/pith/73527J7JDIVSKROCF3WJRJDS4V","download_json":"https://pith.science/pith/73527J7JDIVSKROCF3WJRJDS4V.json","view_paper":"https://pith.science/paper/73527J7J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.15762&json=true","fetch_graph":"https://pith.science/api/pith-number/73527J7JDIVSKROCF3WJRJDS4V/graph.json","fetch_events":"https://pith.science/api/pith-number/73527J7JDIVSKROCF3WJRJDS4V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/73527J7JDIVSKROCF3WJRJDS4V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/73527J7JDIVSKROCF3WJRJDS4V/action/storage_attestation","attest_author":"https://pith.science/pith/73527J7JDIVSKROCF3WJRJDS4V/action/author_attestation","sign_citation":"https://pith.science/pith/73527J7JDIVSKROCF3WJRJDS4V/action/citation_signature","submit_replication":"https://pith.science/pith/73527J7JDIVSKROCF3WJRJDS4V/action/replication_record"}},"created_at":"2026-07-05T09:24:37.050412+00:00","updated_at":"2026-07-05T09:24:37.050412+00:00"}