{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XYEK7QTKBGE5KHEQOBUEULRMCK","short_pith_number":"pith:XYEK7QTK","schema_version":"1.0","canonical_sha256":"be08afc26a0989d51c9070684a2e2c128c46fe1f3a35d6e4933f4cf32b9dd6d7","source":{"kind":"arxiv","id":"2505.17063","version":1},"attestation_state":"computed","paper":{"title":"Synthetic Data RL: Task Definition Is All You Need","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chuanwei Huang, Haofei Yu, Huishuai Zhang, Yiduo Guo, Yikang Shen, Zekai Zhang, Zhen Guo, Zi-ang Wang","submitted_at":"2025-05-18T05:35:13Z","abstract_excerpt":"Reinforcement learning (RL) is a powerful way to adapt foundation models to specialized tasks, but its reliance on large-scale human-labeled data limits broad adoption. We introduce Synthetic Data RL, a simple and general framework that reinforcement fine-tunes models using only synthetic data generated from a task definition. Our method first generates question and answer pairs from the task definition and retrieved documents, then adapts the difficulty of the question based on model solvability, and selects questions using the average pass rate of the model across samples for RL training. On"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17063","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-18T05:35:13Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3411af2244fa1e14f1d1abd3a996b86119d309e189231289efe6a887d753db54","abstract_canon_sha256":"d3eee70e547742c058e289b9acfb765947f442fa2dcf9564a9b0759e02392813"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:46.466439Z","signature_b64":"VopCd5hPp3SJOP7at3raVJqAysvhNaE0kXQIPTGEhsKy7ifwZlQCn/NMd/XiH/E2RuXZe26gMq93RaNF2c0uBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be08afc26a0989d51c9070684a2e2c128c46fe1f3a35d6e4933f4cf32b9dd6d7","last_reissued_at":"2026-07-05T11:07:46.465813Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:46.465813Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Synthetic Data RL: Task Definition Is All You Need","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chuanwei Huang, Haofei Yu, Huishuai Zhang, Yiduo Guo, Yikang Shen, Zekai Zhang, Zhen Guo, Zi-ang Wang","submitted_at":"2025-05-18T05:35:13Z","abstract_excerpt":"Reinforcement learning (RL) is a powerful way to adapt foundation models to specialized tasks, but its reliance on large-scale human-labeled data limits broad adoption. We introduce Synthetic Data RL, a simple and general framework that reinforcement fine-tunes models using only synthetic data generated from a task definition. Our method first generates question and answer pairs from the task definition and retrieved documents, then adapts the difficulty of the question based on model solvability, and selects questions using the average pass rate of the model across samples for RL training. On"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17063","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17063/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17063","created_at":"2026-07-05T11:07:46.465898+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17063v1","created_at":"2026-07-05T11:07:46.465898+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17063","created_at":"2026-07-05T11:07:46.465898+00:00"},{"alias_kind":"pith_short_12","alias_value":"XYEK7QTKBGE5","created_at":"2026-07-05T11:07:46.465898+00:00"},{"alias_kind":"pith_short_16","alias_value":"XYEK7QTKBGE5KHEQ","created_at":"2026-07-05T11:07:46.465898+00:00"},{"alias_kind":"pith_short_8","alias_value":"XYEK7QTK","created_at":"2026-07-05T11:07:46.465898+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":238,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31238","citing_title":"Scaling Multi-Hop Training Data via Graph-Constrained Path Selection","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02547","citing_title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2511.09907","citing_title":"Learning to Pose Problems: Reasoning-Driven and Solver-Adaptive Data Synthesis","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2601.04809","citing_title":"SCALER:Synthetic Scalable Adaptive Learning Environment for Reasoning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08138","citing_title":"DataArc-SynData-Toolkit: A Unified Closed-Loop Framework for Multi-Path, Multimodal, and Multilingual Data Synthesis","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XYEK7QTKBGE5KHEQOBUEULRMCK","json":"https://pith.science/pith/XYEK7QTKBGE5KHEQOBUEULRMCK.json","graph_json":"https://pith.science/api/pith-number/XYEK7QTKBGE5KHEQOBUEULRMCK/graph.json","events_json":"https://pith.science/api/pith-number/XYEK7QTKBGE5KHEQOBUEULRMCK/events.json","paper":"https://pith.science/paper/XYEK7QTK"},"agent_actions":{"view_html":"https://pith.science/pith/XYEK7QTKBGE5KHEQOBUEULRMCK","download_json":"https://pith.science/pith/XYEK7QTKBGE5KHEQOBUEULRMCK.json","view_paper":"https://pith.science/paper/XYEK7QTK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17063&json=true","fetch_graph":"https://pith.science/api/pith-number/XYEK7QTKBGE5KHEQOBUEULRMCK/graph.json","fetch_events":"https://pith.science/api/pith-number/XYEK7QTKBGE5KHEQOBUEULRMCK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XYEK7QTKBGE5KHEQOBUEULRMCK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XYEK7QTKBGE5KHEQOBUEULRMCK/action/storage_attestation","attest_author":"https://pith.science/pith/XYEK7QTKBGE5KHEQOBUEULRMCK/action/author_attestation","sign_citation":"https://pith.science/pith/XYEK7QTKBGE5KHEQOBUEULRMCK/action/citation_signature","submit_replication":"https://pith.science/pith/XYEK7QTKBGE5KHEQOBUEULRMCK/action/replication_record"}},"created_at":"2026-07-05T11:07:46.465898+00:00","updated_at":"2026-07-05T11:07:46.465898+00:00"}