{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VUMJ4IFSLLVFSWOK3XSQZA2YHX","short_pith_number":"pith:VUMJ4IFS","schema_version":"1.0","canonical_sha256":"ad189e20b25aea5959cadde50c83583dd1e4c0ef183362daa02831016edb2182","source":{"kind":"arxiv","id":"2509.07980","version":2},"attestation_state":"computed","paper":{"title":"Parallel-R1: Towards Parallel Thinking via Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chengsong Huang, Dong Yu, Heng Huang, Hongming Zhang, Huiwen Bao, Rui Liu, Runpeng Dai, Tong Zheng, Wenhao Yu, Xiaoyang Wang","submitted_at":"2025-09-09T17:59:35Z","abstract_excerpt":"Parallel thinking has emerged as a novel approach for enhancing the reasoning capabilities of large language models (LLMs) by exploring multiple reasoning paths concurrently. However, activating such capabilities through training remains challenging, as existing methods predominantly rely on supervised fine-tuning (SFT) over synthetic data, which encourages teacher-forced imitation rather than exploration and generalization. Different from them, we propose \\textbf{Parallel-R1}, the first reinforcement learning (RL) framework that enables parallel thinking behaviors for complex real-world reaso"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.07980","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-09-09T17:59:35Z","cross_cats_sorted":[],"title_canon_sha256":"4fb5fabbc768165f7994b8657a38e698dc35725d846e98f62f2a80845cec47ef","abstract_canon_sha256":"7e3188282779ed5483bd563c2fde6cbcef07e795e7e47bebe7a70b4f97f1aecb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:11:00.377961Z","signature_b64":"EmPhuRBwfj1wir77z2POVjkeeQHYOZ5JyxzXkzhBaAZfLQ2e8l8Vb6RKxS9YuiB+AQYXsfNFJhBoUpwKP1QiDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ad189e20b25aea5959cadde50c83583dd1e4c0ef183362daa02831016edb2182","last_reissued_at":"2026-07-05T12:11:00.377416Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:11:00.377416Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Parallel-R1: Towards Parallel Thinking via Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chengsong Huang, Dong Yu, Heng Huang, Hongming Zhang, Huiwen Bao, Rui Liu, Runpeng Dai, Tong Zheng, Wenhao Yu, Xiaoyang Wang","submitted_at":"2025-09-09T17:59:35Z","abstract_excerpt":"Parallel thinking has emerged as a novel approach for enhancing the reasoning capabilities of large language models (LLMs) by exploring multiple reasoning paths concurrently. However, activating such capabilities through training remains challenging, as existing methods predominantly rely on supervised fine-tuning (SFT) over synthetic data, which encourages teacher-forced imitation rather than exploration and generalization. Different from them, we propose \\textbf{Parallel-R1}, the first reinforcement learning (RL) framework that enables parallel thinking behaviors for complex real-world reaso"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.07980","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.07980/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.07980","created_at":"2026-07-05T12:11:00.377475+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.07980v2","created_at":"2026-07-05T12:11:00.377475+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.07980","created_at":"2026-07-05T12:11:00.377475+00:00"},{"alias_kind":"pith_short_12","alias_value":"VUMJ4IFSLLVF","created_at":"2026-07-05T12:11:00.377475+00:00"},{"alias_kind":"pith_short_16","alias_value":"VUMJ4IFSLLVFSWOK","created_at":"2026-07-05T12:11:00.377475+00:00"},{"alias_kind":"pith_short_8","alias_value":"VUMJ4IFS","created_at":"2026-07-05T12:11:00.377475+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03102","citing_title":"Small RL Controller, Large Language Model: RL-Guided Adaptive Sampling for Test-Time Scaling","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07461","citing_title":"Native Parallel Reasoner: Reasoning in Parallelism via Self-Distilled Reinforcement Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2512.17445","citing_title":"LangDriveCTRL: Natural Language Controllable Driving Scene Editing with Multi-modal Agents","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21619","citing_title":"On the Overscaling Curse of Parallel Thinking: System Efficacy Contradicts Sample Efficiency","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15529","citing_title":"LACE: Lattice Attention for Cross-thread Exploration","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09262","citing_title":"Reinforcing Multimodal Reasoning Against Visual Degradation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09269","citing_title":"DeltaRubric: Generative Multimodal Reward Modeling via Joint Planning and Verification","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04330","citing_title":"The Scaling Properties of Implicit Deductive Reasoning in Transformers","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06914","citing_title":"Regulating Branch Parallelism in LLM Serving","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15529","citing_title":"LACE: Lattice Attention for Cross-thread Exploration","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05868","citing_title":"Understanding Performance Gap Between Parallel and Sequential Sampling in Large Reasoning Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15529","citing_title":"LACE: Lattice Attention for Cross-thread Exploration","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18493","citing_title":"Too Correct to Learn: Reinforcement Learning on Saturated Reasoning Data","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VUMJ4IFSLLVFSWOK3XSQZA2YHX","json":"https://pith.science/pith/VUMJ4IFSLLVFSWOK3XSQZA2YHX.json","graph_json":"https://pith.science/api/pith-number/VUMJ4IFSLLVFSWOK3XSQZA2YHX/graph.json","events_json":"https://pith.science/api/pith-number/VUMJ4IFSLLVFSWOK3XSQZA2YHX/events.json","paper":"https://pith.science/paper/VUMJ4IFS"},"agent_actions":{"view_html":"https://pith.science/pith/VUMJ4IFSLLVFSWOK3XSQZA2YHX","download_json":"https://pith.science/pith/VUMJ4IFSLLVFSWOK3XSQZA2YHX.json","view_paper":"https://pith.science/paper/VUMJ4IFS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.07980&json=true","fetch_graph":"https://pith.science/api/pith-number/VUMJ4IFSLLVFSWOK3XSQZA2YHX/graph.json","fetch_events":"https://pith.science/api/pith-number/VUMJ4IFSLLVFSWOK3XSQZA2YHX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VUMJ4IFSLLVFSWOK3XSQZA2YHX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VUMJ4IFSLLVFSWOK3XSQZA2YHX/action/storage_attestation","attest_author":"https://pith.science/pith/VUMJ4IFSLLVFSWOK3XSQZA2YHX/action/author_attestation","sign_citation":"https://pith.science/pith/VUMJ4IFSLLVFSWOK3XSQZA2YHX/action/citation_signature","submit_replication":"https://pith.science/pith/VUMJ4IFSLLVFSWOK3XSQZA2YHX/action/replication_record"}},"created_at":"2026-07-05T12:11:00.377475+00:00","updated_at":"2026-07-05T12:11:00.377475+00:00"}