{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YUNIGC5C65JZLUKCPTCCGK2QXR","short_pith_number":"pith:YUNIGC5C","schema_version":"1.0","canonical_sha256":"c51a830ba2f75395d1427cc4232b50bc46ae0e4b5c12d8d9056fe78790e5aeb3","source":{"kind":"arxiv","id":"2505.14157","version":2},"attestation_state":"computed","paper":{"title":"Prior Prompt Engineering for Reinforcement Fine-Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Kunat Pipatanakul, Pittawat Taveekitworachai, Potsawee Manakul, Sarana Nutanong","submitted_at":"2025-05-20T10:05:11Z","abstract_excerpt":"This paper investigates prior prompt engineering (pPE) in the context of reinforcement fine-tuning (RFT), where language models (LMs) are incentivized to exhibit behaviors that maximize performance through reward signals. While existing RFT research has primarily focused on algorithms, reward shaping, and data curation, the design of the prior prompt--the instructions prepended to queries during training to elicit behaviors such as step-by-step reasoning--remains underexplored. We investigate whether different pPE approaches can guide LMs to internalize distinct behaviors after RFT. Inspired b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14157","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-20T10:05:11Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"01c0c5fff661059bf21b4f2030e79193adf29ce7a25cd3a45d71563a314f56a1","abstract_canon_sha256":"eeebfc5ba25f5eb10bd7d772cad037bbe598bc5c6213475f8a21716f81baa282"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:08:03.013890Z","signature_b64":"1qLqPjTNsJEAKF0XyBeULesmK7ao7gam4N5ZK5ET9NjC1IqiZVoiFBURTH446/yIxbErJJDntaU4/hk6g0VNAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c51a830ba2f75395d1427cc4232b50bc46ae0e4b5c12d8d9056fe78790e5aeb3","last_reissued_at":"2026-07-05T12:08:03.013355Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:08:03.013355Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prior Prompt Engineering for Reinforcement Fine-Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Kunat Pipatanakul, Pittawat Taveekitworachai, Potsawee Manakul, Sarana Nutanong","submitted_at":"2025-05-20T10:05:11Z","abstract_excerpt":"This paper investigates prior prompt engineering (pPE) in the context of reinforcement fine-tuning (RFT), where language models (LMs) are incentivized to exhibit behaviors that maximize performance through reward signals. While existing RFT research has primarily focused on algorithms, reward shaping, and data curation, the design of the prior prompt--the instructions prepended to queries during training to elicit behaviors such as step-by-step reasoning--remains underexplored. We investigate whether different pPE approaches can guide LMs to internalize distinct behaviors after RFT. Inspired b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14157","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14157/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14157","created_at":"2026-07-05T12:08:03.013418+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14157v2","created_at":"2026-07-05T12:08:03.013418+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14157","created_at":"2026-07-05T12:08:03.013418+00:00"},{"alias_kind":"pith_short_12","alias_value":"YUNIGC5C65JZ","created_at":"2026-07-05T12:08:03.013418+00:00"},{"alias_kind":"pith_short_16","alias_value":"YUNIGC5C65JZLUKC","created_at":"2026-07-05T12:08:03.013418+00:00"},{"alias_kind":"pith_short_8","alias_value":"YUNIGC5C","created_at":"2026-07-05T12:08:03.013418+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YUNIGC5C65JZLUKCPTCCGK2QXR","json":"https://pith.science/pith/YUNIGC5C65JZLUKCPTCCGK2QXR.json","graph_json":"https://pith.science/api/pith-number/YUNIGC5C65JZLUKCPTCCGK2QXR/graph.json","events_json":"https://pith.science/api/pith-number/YUNIGC5C65JZLUKCPTCCGK2QXR/events.json","paper":"https://pith.science/paper/YUNIGC5C"},"agent_actions":{"view_html":"https://pith.science/pith/YUNIGC5C65JZLUKCPTCCGK2QXR","download_json":"https://pith.science/pith/YUNIGC5C65JZLUKCPTCCGK2QXR.json","view_paper":"https://pith.science/paper/YUNIGC5C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14157&json=true","fetch_graph":"https://pith.science/api/pith-number/YUNIGC5C65JZLUKCPTCCGK2QXR/graph.json","fetch_events":"https://pith.science/api/pith-number/YUNIGC5C65JZLUKCPTCCGK2QXR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YUNIGC5C65JZLUKCPTCCGK2QXR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YUNIGC5C65JZLUKCPTCCGK2QXR/action/storage_attestation","attest_author":"https://pith.science/pith/YUNIGC5C65JZLUKCPTCCGK2QXR/action/author_attestation","sign_citation":"https://pith.science/pith/YUNIGC5C65JZLUKCPTCCGK2QXR/action/citation_signature","submit_replication":"https://pith.science/pith/YUNIGC5C65JZLUKCPTCCGK2QXR/action/replication_record"}},"created_at":"2026-07-05T12:08:03.013418+00:00","updated_at":"2026-07-05T12:08:03.013418+00:00"}