{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CQ6RLKHHSAKVLOADYU7S2FLXOO","short_pith_number":"pith:CQ6RLKHH","schema_version":"1.0","canonical_sha256":"143d15a8e7901555b803c53f2d157773abeb952fc17f5fd273bc874d1c2ae5fe","source":{"kind":"arxiv","id":"2402.11711","version":2},"attestation_state":"computed","paper":{"title":"MORL-Prompt: An Empirical Analysis of Multi-Objective Reinforcement Learning for Discrete Prompt Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dheeraj Mekala, Rose Yu, Taylor Berg-Kirkpatrick, Yasaman Jafari","submitted_at":"2024-02-18T21:25:09Z","abstract_excerpt":"RL-based techniques can be employed to search for prompts that, when fed into a target language model, maximize a set of user-specified reward functions. However, in many target applications, the natural reward functions are in tension with one another -- for example, content preservation vs. style matching in style transfer tasks. Current techniques focus on maximizing the average of reward functions, which does not necessarily lead to prompts that achieve balance across rewards -- an issue that has been well-studied in the multi-objective and robust optimization literature. In this paper, we"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.11711","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-18T21:25:09Z","cross_cats_sorted":[],"title_canon_sha256":"4f18ce4228aa19a02484ef568a76df3cc551e5eb3c01884b501db80b40597073","abstract_canon_sha256":"ed67600da67ec8728aa4dec43305aa89558f042daf49f86f1a35978a82473267"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:11.279533Z","signature_b64":"QHnkOgCxfytTbrYXfY8/voM3hyGtLfDME/8g3nfKsw/GjfNXO8vi+ezznRrFSSW1p/mTmgywZXI7tSWcptfVCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"143d15a8e7901555b803c53f2d157773abeb952fc17f5fd273bc874d1c2ae5fe","last_reissued_at":"2026-07-05T11:18:11.279062Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:11.279062Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MORL-Prompt: An Empirical Analysis of Multi-Objective Reinforcement Learning for Discrete Prompt Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dheeraj Mekala, Rose Yu, Taylor Berg-Kirkpatrick, Yasaman Jafari","submitted_at":"2024-02-18T21:25:09Z","abstract_excerpt":"RL-based techniques can be employed to search for prompts that, when fed into a target language model, maximize a set of user-specified reward functions. However, in many target applications, the natural reward functions are in tension with one another -- for example, content preservation vs. style matching in style transfer tasks. Current techniques focus on maximizing the average of reward functions, which does not necessarily lead to prompts that achieve balance across rewards -- an issue that has been well-studied in the multi-objective and robust optimization literature. In this paper, we"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.11711","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.11711/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.11711","created_at":"2026-07-05T11:18:11.279117+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.11711v2","created_at":"2026-07-05T11:18:11.279117+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.11711","created_at":"2026-07-05T11:18:11.279117+00:00"},{"alias_kind":"pith_short_12","alias_value":"CQ6RLKHHSAKV","created_at":"2026-07-05T11:18:11.279117+00:00"},{"alias_kind":"pith_short_16","alias_value":"CQ6RLKHHSAKVLOAD","created_at":"2026-07-05T11:18:11.279117+00:00"},{"alias_kind":"pith_short_8","alias_value":"CQ6RLKHH","created_at":"2026-07-05T11:18:11.279117+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20408","citing_title":"Spectral Souping: A Unified Framework for Online Preference Alignment","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CQ6RLKHHSAKVLOADYU7S2FLXOO","json":"https://pith.science/pith/CQ6RLKHHSAKVLOADYU7S2FLXOO.json","graph_json":"https://pith.science/api/pith-number/CQ6RLKHHSAKVLOADYU7S2FLXOO/graph.json","events_json":"https://pith.science/api/pith-number/CQ6RLKHHSAKVLOADYU7S2FLXOO/events.json","paper":"https://pith.science/paper/CQ6RLKHH"},"agent_actions":{"view_html":"https://pith.science/pith/CQ6RLKHHSAKVLOADYU7S2FLXOO","download_json":"https://pith.science/pith/CQ6RLKHHSAKVLOADYU7S2FLXOO.json","view_paper":"https://pith.science/paper/CQ6RLKHH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.11711&json=true","fetch_graph":"https://pith.science/api/pith-number/CQ6RLKHHSAKVLOADYU7S2FLXOO/graph.json","fetch_events":"https://pith.science/api/pith-number/CQ6RLKHHSAKVLOADYU7S2FLXOO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CQ6RLKHHSAKVLOADYU7S2FLXOO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CQ6RLKHHSAKVLOADYU7S2FLXOO/action/storage_attestation","attest_author":"https://pith.science/pith/CQ6RLKHHSAKVLOADYU7S2FLXOO/action/author_attestation","sign_citation":"https://pith.science/pith/CQ6RLKHHSAKVLOADYU7S2FLXOO/action/citation_signature","submit_replication":"https://pith.science/pith/CQ6RLKHHSAKVLOADYU7S2FLXOO/action/replication_record"}},"created_at":"2026-07-05T11:18:11.279117+00:00","updated_at":"2026-07-05T11:18:11.279117+00:00"}