{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J74QOVOAZ7IIZFLAGZKGX7TGVU","short_pith_number":"pith:J74QOVOA","schema_version":"1.0","canonical_sha256":"4ff90755c0cfd08c956036546bfe66ad1ade736b2e7b57474211475d7bcf283b","source":{"kind":"arxiv","id":"2506.16024","version":1},"attestation_state":"computed","paper":{"title":"From General to Targeted Rewards: Surpassing GPT-4 in Open-Ended Long-Context Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Irwin King, Jiele Wu, Minda Hu, Wenqian Cui, Yifei Zhang, Yufei Wang, Zhihan Guo","submitted_at":"2025-06-19T04:44:34Z","abstract_excerpt":"Current research on long-form context in Large Language Models (LLMs) primarily focuses on the understanding of long-contexts, the Open-ended Long Text Generation (Open-LTG) remains insufficiently explored. Training a long-context generation model requires curation of gold standard reference data, which is typically nonexistent for informative Open-LTG tasks. However, previous methods only utilize general assessments as reward signals, which limits accuracy. To bridge this gap, we introduce ProxyReward, an innovative reinforcement learning (RL) based framework, which includes a dataset and a r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.16024","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-19T04:44:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fa471df6fcfa41de45281f8df7cbf05e9ea3d9772a2681080f522b092df79ffd","abstract_canon_sha256":"464a53465d51c47615eef1b9a5dda2ebb371396f2a9bbf1221338d5bed53e2a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:24:28.430365Z","signature_b64":"iqvMmglKazqbboZfp4eLwGIEK4+KjZvIVig92hrIZN65QQMmVsGe1IGgiDIjBGQ6zqf6wT7vLemdsMUGq2MQAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ff90755c0cfd08c956036546bfe66ad1ade736b2e7b57474211475d7bcf283b","last_reissued_at":"2026-07-05T11:24:28.429875Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:24:28.429875Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From General to Targeted Rewards: Surpassing GPT-4 in Open-Ended Long-Context Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Irwin King, Jiele Wu, Minda Hu, Wenqian Cui, Yifei Zhang, Yufei Wang, Zhihan Guo","submitted_at":"2025-06-19T04:44:34Z","abstract_excerpt":"Current research on long-form context in Large Language Models (LLMs) primarily focuses on the understanding of long-contexts, the Open-ended Long Text Generation (Open-LTG) remains insufficiently explored. Training a long-context generation model requires curation of gold standard reference data, which is typically nonexistent for informative Open-LTG tasks. However, previous methods only utilize general assessments as reward signals, which limits accuracy. To bridge this gap, we introduce ProxyReward, an innovative reinforcement learning (RL) based framework, which includes a dataset and a r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.16024","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.16024/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.16024","created_at":"2026-07-05T11:24:28.429933+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.16024v1","created_at":"2026-07-05T11:24:28.429933+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.16024","created_at":"2026-07-05T11:24:28.429933+00:00"},{"alias_kind":"pith_short_12","alias_value":"J74QOVOAZ7II","created_at":"2026-07-05T11:24:28.429933+00:00"},{"alias_kind":"pith_short_16","alias_value":"J74QOVOAZ7IIZFLA","created_at":"2026-07-05T11:24:28.429933+00:00"},{"alias_kind":"pith_short_8","alias_value":"J74QOVOA","created_at":"2026-07-05T11:24:28.429933+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":171,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":176,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07465","citing_title":"SEIF: Self-Evolving Reinforcement Learning for Instruction Following","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J74QOVOAZ7IIZFLAGZKGX7TGVU","json":"https://pith.science/pith/J74QOVOAZ7IIZFLAGZKGX7TGVU.json","graph_json":"https://pith.science/api/pith-number/J74QOVOAZ7IIZFLAGZKGX7TGVU/graph.json","events_json":"https://pith.science/api/pith-number/J74QOVOAZ7IIZFLAGZKGX7TGVU/events.json","paper":"https://pith.science/paper/J74QOVOA"},"agent_actions":{"view_html":"https://pith.science/pith/J74QOVOAZ7IIZFLAGZKGX7TGVU","download_json":"https://pith.science/pith/J74QOVOAZ7IIZFLAGZKGX7TGVU.json","view_paper":"https://pith.science/paper/J74QOVOA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.16024&json=true","fetch_graph":"https://pith.science/api/pith-number/J74QOVOAZ7IIZFLAGZKGX7TGVU/graph.json","fetch_events":"https://pith.science/api/pith-number/J74QOVOAZ7IIZFLAGZKGX7TGVU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J74QOVOAZ7IIZFLAGZKGX7TGVU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J74QOVOAZ7IIZFLAGZKGX7TGVU/action/storage_attestation","attest_author":"https://pith.science/pith/J74QOVOAZ7IIZFLAGZKGX7TGVU/action/author_attestation","sign_citation":"https://pith.science/pith/J74QOVOAZ7IIZFLAGZKGX7TGVU/action/citation_signature","submit_replication":"https://pith.science/pith/J74QOVOAZ7IIZFLAGZKGX7TGVU/action/replication_record"}},"created_at":"2026-07-05T11:24:28.429933+00:00","updated_at":"2026-07-05T11:24:28.429933+00:00"}