{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J5UDUKID5O33XT3I4MWTG7ULDL","short_pith_number":"pith:J5UDUKID","schema_version":"1.0","canonical_sha256":"4f683a2903ebb7bbcf68e32d337e8b1ac75aa868f5ac2c71cf1770bc994179e5","source":{"kind":"arxiv","id":"2508.20722","version":1},"attestation_state":"computed","paper":{"title":"rStar2-Agent: Agentic Reasoning Technical Report","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bingcheng Dong, Bowen Zhang, Buze Zhang, Fan Yang, Li Lyna Zhang, Mao Yang, Ning Shang, Scarlett Li, Weijiang Xu, Xinyu Guan, XuDong Zhou, Yifei Liu, Ying Xin, Yi Zhu, Ziming Miao","submitted_at":"2025-08-28T12:45:25Z","abstract_excerpt":"We introduce rStar2-Agent, a 14B math reasoning model trained with agentic reinforcement learning to achieve frontier-level performance. Beyond current long CoT, the model demonstrates advanced cognitive behaviors, such as thinking carefully before using Python coding tools and reflecting on code execution feedback to autonomously explore, verify, and refine intermediate steps in complex problem-solving. This capability is enabled through three key innovations that makes agentic RL effective at scale: (i) an efficient RL infrastructure with a reliable Python code environment that supports high"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.20722","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-28T12:45:25Z","cross_cats_sorted":[],"title_canon_sha256":"bcddcd8b020addd6fbe514ac58e3f91e7f5120cb25c4b711fe6d200db51bdfe6","abstract_canon_sha256":"f393ddf1dcfb4d47fb280118d000a86224443443939f782024e8222be074e2f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:01:09.651018Z","signature_b64":"/x0MhX6N+J2juAQNQkfOstT2aSoLt9NRIYzl4e4P9fq1BVLWnYpw4Y4VDhLAuAFEizgxpd5Rc0LfAPtpmybCDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f683a2903ebb7bbcf68e32d337e8b1ac75aa868f5ac2c71cf1770bc994179e5","last_reissued_at":"2026-07-05T12:01:09.650534Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:01:09.650534Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"rStar2-Agent: Agentic Reasoning Technical Report","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bingcheng Dong, Bowen Zhang, Buze Zhang, Fan Yang, Li Lyna Zhang, Mao Yang, Ning Shang, Scarlett Li, Weijiang Xu, Xinyu Guan, XuDong Zhou, Yifei Liu, Ying Xin, Yi Zhu, Ziming Miao","submitted_at":"2025-08-28T12:45:25Z","abstract_excerpt":"We introduce rStar2-Agent, a 14B math reasoning model trained with agentic reinforcement learning to achieve frontier-level performance. Beyond current long CoT, the model demonstrates advanced cognitive behaviors, such as thinking carefully before using Python coding tools and reflecting on code execution feedback to autonomously explore, verify, and refine intermediate steps in complex problem-solving. This capability is enabled through three key innovations that makes agentic RL effective at scale: (i) an efficient RL infrastructure with a reliable Python code environment that supports high"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.20722","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.20722/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.20722","created_at":"2026-07-05T12:01:09.650590+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.20722v1","created_at":"2026-07-05T12:01:09.650590+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.20722","created_at":"2026-07-05T12:01:09.650590+00:00"},{"alias_kind":"pith_short_12","alias_value":"J5UDUKID5O33","created_at":"2026-07-05T12:01:09.650590+00:00"},{"alias_kind":"pith_short_16","alias_value":"J5UDUKID5O33XT3I","created_at":"2026-07-05T12:01:09.650590+00:00"},{"alias_kind":"pith_short_8","alias_value":"J5UDUKID","created_at":"2026-07-05T12:01:09.650590+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26620","citing_title":"Discovering Millions of Interpretable Features with Sparse Autoencoders","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03762","citing_title":"Tool-Aware Optimization with Entropy Guidance for Efficient Agentic Reinforcement Learning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24693","citing_title":"CP-Agent: A Calibrated Risk-Controlled Agent for Feedback-Driven Competitive Programming","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28774","citing_title":"Agent Explorative Policy Optimization for Multimodal Agentic Reasoning","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18500","citing_title":"Implicit Hierarchical GRPO: Decoupling Tool Invocation from Execution for Tool-Integrated Mathematical Reasoning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02547","citing_title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07237","citing_title":"Teaching Language Models to Think in Code","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23838","citing_title":"JigsawRL: Assembling RL Pipelines for Efficient LLM Post-Training","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06326","citing_title":"Teaching Thinking Models to Reason with Tools: A Full-Pipeline Recipe for Tool-Integrated Reasoning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04719","citing_title":"Every Step Counts: Step-Level Credit Assignment for Tool-Integrated Text-to-SQL","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18936","citing_title":"Fine-Tuning Small Reasoning Models for Quantum Field Theory","ref_index":224,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07237","citing_title":"Teaching Language Models to Think in Code","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06636","citing_title":"SHAPE: Stage-aware Hierarchical Advantage via Potential Estimation for LLM Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02178","citing_title":"T$^2$PO: Uncertainty-Guided Exploration Control for Stable Multi-Turn Agentic Reinforcement Learning","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J5UDUKID5O33XT3I4MWTG7ULDL","json":"https://pith.science/pith/J5UDUKID5O33XT3I4MWTG7ULDL.json","graph_json":"https://pith.science/api/pith-number/J5UDUKID5O33XT3I4MWTG7ULDL/graph.json","events_json":"https://pith.science/api/pith-number/J5UDUKID5O33XT3I4MWTG7ULDL/events.json","paper":"https://pith.science/paper/J5UDUKID"},"agent_actions":{"view_html":"https://pith.science/pith/J5UDUKID5O33XT3I4MWTG7ULDL","download_json":"https://pith.science/pith/J5UDUKID5O33XT3I4MWTG7ULDL.json","view_paper":"https://pith.science/paper/J5UDUKID","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.20722&json=true","fetch_graph":"https://pith.science/api/pith-number/J5UDUKID5O33XT3I4MWTG7ULDL/graph.json","fetch_events":"https://pith.science/api/pith-number/J5UDUKID5O33XT3I4MWTG7ULDL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J5UDUKID5O33XT3I4MWTG7ULDL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J5UDUKID5O33XT3I4MWTG7ULDL/action/storage_attestation","attest_author":"https://pith.science/pith/J5UDUKID5O33XT3I4MWTG7ULDL/action/author_attestation","sign_citation":"https://pith.science/pith/J5UDUKID5O33XT3I4MWTG7ULDL/action/citation_signature","submit_replication":"https://pith.science/pith/J5UDUKID5O33XT3I4MWTG7ULDL/action/replication_record"}},"created_at":"2026-07-05T12:01:09.650590+00:00","updated_at":"2026-07-05T12:01:09.650590+00:00"}