{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WO3YSO3W7CLQC3473N5VG32HRH","short_pith_number":"pith:WO3YSO3W","schema_version":"1.0","canonical_sha256":"b3b7893b76f897016f9fdb7b536f4789e4acf766408a228ee11a95071a48d68c","source":{"kind":"arxiv","id":"2503.20194","version":1},"attestation_state":"computed","paper":{"title":"GAPO: Learning Preferential Prompt through Generative Adversarial Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hongwei Feng, Suhang Zheng, Tao Wang, Tianyu Li, Xiaoran Shi, Xingzhou Chen, Yanghua Xiao, Zhouhong Gu","submitted_at":"2025-03-26T03:37:52Z","abstract_excerpt":"Recent advances in large language models have highlighted the critical need for precise control over model outputs through predefined constraints. While existing methods attempt to achieve this through either direct instruction-response synthesis or preferential response optimization, they often struggle with constraint understanding and adaptation. This limitation becomes particularly evident when handling fine-grained constraints, leading to either hallucination or brittle performance. We introduce Generative Adversarial Policy Optimization (GAPO), a novel framework that combines GAN-based t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.20194","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-26T03:37:52Z","cross_cats_sorted":[],"title_canon_sha256":"12b192545de11c052e37334bfa073609b55c51cd32a8ac5aac6a4b47e34307ab","abstract_canon_sha256":"2c2237fcceb8f048622a1b875f1849b18cb6139060d91bf142851cd65af45de5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:39:34.371413Z","signature_b64":"Rhf77+eA4F5x0Q2IjUSObvbXcmZBNG/XML7tbAZvlOG/IVJAMnP9V8JEm3b06b1eSh3LmGVMORAUoJ9GCBAaDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b3b7893b76f897016f9fdb7b536f4789e4acf766408a228ee11a95071a48d68c","last_reissued_at":"2026-07-05T10:39:34.370885Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:39:34.370885Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GAPO: Learning Preferential Prompt through Generative Adversarial Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hongwei Feng, Suhang Zheng, Tao Wang, Tianyu Li, Xiaoran Shi, Xingzhou Chen, Yanghua Xiao, Zhouhong Gu","submitted_at":"2025-03-26T03:37:52Z","abstract_excerpt":"Recent advances in large language models have highlighted the critical need for precise control over model outputs through predefined constraints. While existing methods attempt to achieve this through either direct instruction-response synthesis or preferential response optimization, they often struggle with constraint understanding and adaptation. This limitation becomes particularly evident when handling fine-grained constraints, leading to either hallucination or brittle performance. We introduce Generative Adversarial Policy Optimization (GAPO), a novel framework that combines GAN-based t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.20194","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.20194/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.20194","created_at":"2026-07-05T10:39:34.370951+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.20194v1","created_at":"2026-07-05T10:39:34.370951+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.20194","created_at":"2026-07-05T10:39:34.370951+00:00"},{"alias_kind":"pith_short_12","alias_value":"WO3YSO3W7CLQ","created_at":"2026-07-05T10:39:34.370951+00:00"},{"alias_kind":"pith_short_16","alias_value":"WO3YSO3W7CLQC347","created_at":"2026-07-05T10:39:34.370951+00:00"},{"alias_kind":"pith_short_8","alias_value":"WO3YSO3W","created_at":"2026-07-05T10:39:34.370951+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30789","citing_title":"Smaller Models are Natural Explorers for Policy-Level Diversity in GRPO","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02988","citing_title":"Self-Optimizing Multi-Agent Systems for Deep Research","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06036","citing_title":"Optimal Transport for LLM Reward Modeling from Noisy Preference","ref_index":251,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WO3YSO3W7CLQC3473N5VG32HRH","json":"https://pith.science/pith/WO3YSO3W7CLQC3473N5VG32HRH.json","graph_json":"https://pith.science/api/pith-number/WO3YSO3W7CLQC3473N5VG32HRH/graph.json","events_json":"https://pith.science/api/pith-number/WO3YSO3W7CLQC3473N5VG32HRH/events.json","paper":"https://pith.science/paper/WO3YSO3W"},"agent_actions":{"view_html":"https://pith.science/pith/WO3YSO3W7CLQC3473N5VG32HRH","download_json":"https://pith.science/pith/WO3YSO3W7CLQC3473N5VG32HRH.json","view_paper":"https://pith.science/paper/WO3YSO3W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.20194&json=true","fetch_graph":"https://pith.science/api/pith-number/WO3YSO3W7CLQC3473N5VG32HRH/graph.json","fetch_events":"https://pith.science/api/pith-number/WO3YSO3W7CLQC3473N5VG32HRH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WO3YSO3W7CLQC3473N5VG32HRH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WO3YSO3W7CLQC3473N5VG32HRH/action/storage_attestation","attest_author":"https://pith.science/pith/WO3YSO3W7CLQC3473N5VG32HRH/action/author_attestation","sign_citation":"https://pith.science/pith/WO3YSO3W7CLQC3473N5VG32HRH/action/citation_signature","submit_replication":"https://pith.science/pith/WO3YSO3W7CLQC3473N5VG32HRH/action/replication_record"}},"created_at":"2026-07-05T10:39:34.370951+00:00","updated_at":"2026-07-05T10:39:34.370951+00:00"}