{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZU2Y5F6BHIT4UHFAK6L7AZ72VP","short_pith_number":"pith:ZU2Y5F6B","schema_version":"1.0","canonical_sha256":"cd358e97c13a27ca1ca05797f067faabf1915e30f6cf33165105b4226ed7ea1c","source":{"kind":"arxiv","id":"2508.17445","version":1},"attestation_state":"computed","paper":{"title":"TreePO: Bridging the Gap of Policy Optimization and Efficacy and Inference Efficiency with Heuristic Tree-based Modeling","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chenghua Lin, Ge Zhang, Jian Yang, Qian Liu, Qingshui Gu, Shuyue Guo, Tianshun Xing, Tianyu Zheng, Wangchunshu Zhou, Wei Shen, Wenhao Huang, Xingwei Qu, Xin Zhou, Yizhi Li, Zheng Zhang, Zhoufutu Wen, Ziniu Li","submitted_at":"2025-08-24T16:52:37Z","abstract_excerpt":"Recent advancements in aligning large language models via reinforcement learning have achieved remarkable gains in solving complex reasoning problems, but at the cost of expensive on-policy rollouts and limited exploration of diverse reasoning paths. In this work, we introduce TreePO, involving a self-guided rollout algorithm that views sequence generation as a tree-structured searching process. Composed of dynamic tree sampling policy and fixed-length segment decoding, TreePO leverages local uncertainty to warrant additional branches. By amortizing computation across common prefixes and pruni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.17445","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-24T16:52:37Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"36adbca1dd0761f7af141be1cff02afe16fc3a1575557a00f9401583f9fd6b22","abstract_canon_sha256":"d988675c28f314ec8e4ffbcb31278a402427f5ecd7a4952691be402b39956301"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:58:41.654105Z","signature_b64":"w8DBBg8EU7fcZ6LK4RtsIWFpR8wH0jedPKCrTYgcyDxA2Ic+LCxoNC+SaLPCy9PCYXFSxRLwRbVnM+iq4eMDDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd358e97c13a27ca1ca05797f067faabf1915e30f6cf33165105b4226ed7ea1c","last_reissued_at":"2026-07-05T11:58:41.653627Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:58:41.653627Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TreePO: Bridging the Gap of Policy Optimization and Efficacy and Inference Efficiency with Heuristic Tree-based Modeling","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chenghua Lin, Ge Zhang, Jian Yang, Qian Liu, Qingshui Gu, Shuyue Guo, Tianshun Xing, Tianyu Zheng, Wangchunshu Zhou, Wei Shen, Wenhao Huang, Xingwei Qu, Xin Zhou, Yizhi Li, Zheng Zhang, Zhoufutu Wen, Ziniu Li","submitted_at":"2025-08-24T16:52:37Z","abstract_excerpt":"Recent advancements in aligning large language models via reinforcement learning have achieved remarkable gains in solving complex reasoning problems, but at the cost of expensive on-policy rollouts and limited exploration of diverse reasoning paths. In this work, we introduce TreePO, involving a self-guided rollout algorithm that views sequence generation as a tree-structured searching process. Composed of dynamic tree sampling policy and fixed-length segment decoding, TreePO leverages local uncertainty to warrant additional branches. By amortizing computation across common prefixes and pruni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.17445","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.17445/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.17445","created_at":"2026-07-05T11:58:41.653684+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.17445v1","created_at":"2026-07-05T11:58:41.653684+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.17445","created_at":"2026-07-05T11:58:41.653684+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZU2Y5F6BHIT4","created_at":"2026-07-05T11:58:41.653684+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZU2Y5F6BHIT4UHFA","created_at":"2026-07-05T11:58:41.653684+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZU2Y5F6B","created_at":"2026-07-05T11:58:41.653684+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25451","citing_title":"Learning with a Single Rollout via Monte Carlo Pass@k Critic","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25354","citing_title":"Efficient and Trainable Language Model Test-Time Scaling via Local Branch Routing","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18089","citing_title":"From Reasoning Traces to Reusable Modules: Understanding Compositional Generalization in Language Model Reasoning","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25354","citing_title":"Efficient and Trainable Language Model Test-Time Scaling via Local Branch Routing","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19425","citing_title":"When to Stop Reusing: Dynamic Gradient Gating for Sample-Efficient RLVR","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27859","citing_title":"Rethinking Agentic Reinforcement Learning In Large Language Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02547","citing_title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2511.00413","citing_title":"Tree Training: Accelerating Agentic LLMs Training via Shared Prefix Reuse","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":291,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21619","citing_title":"On the Overscaling Curse of Parallel Thinking: System Efficacy Contradicts Sample Efficiency","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11922","citing_title":"StepCodeReasoner: Aligning Code Reasoning with Stepwise Execution Traces via Reinforcement Learning","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27859","citing_title":"Rethinking Agentic Reinforcement Learning In Large Language Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27859","citing_title":"Rethinking Agentic Reinforcement Learning In Large Language Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04811","citing_title":"Tree-based Credit Assignment for Multi-Agent Memory System","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07353","citing_title":"Confidence-Aware Alignment Makes Reasoning LLMs More Reliable","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02913","citing_title":"Generate, Filter, Control, Replay: A Comprehensive Survey of Rollout Strategies for LLM Reinforcement Learning","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14564","citing_title":"MARS$^2$: Scaling Multi-Agent Tree Search via Reinforcement Learning for Code Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18292","citing_title":"Agent-World: Scaling Real-World Environment Synthesis for Evolving General Agent Intelligence","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZU2Y5F6BHIT4UHFAK6L7AZ72VP","json":"https://pith.science/pith/ZU2Y5F6BHIT4UHFAK6L7AZ72VP.json","graph_json":"https://pith.science/api/pith-number/ZU2Y5F6BHIT4UHFAK6L7AZ72VP/graph.json","events_json":"https://pith.science/api/pith-number/ZU2Y5F6BHIT4UHFAK6L7AZ72VP/events.json","paper":"https://pith.science/paper/ZU2Y5F6B"},"agent_actions":{"view_html":"https://pith.science/pith/ZU2Y5F6BHIT4UHFAK6L7AZ72VP","download_json":"https://pith.science/pith/ZU2Y5F6BHIT4UHFAK6L7AZ72VP.json","view_paper":"https://pith.science/paper/ZU2Y5F6B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.17445&json=true","fetch_graph":"https://pith.science/api/pith-number/ZU2Y5F6BHIT4UHFAK6L7AZ72VP/graph.json","fetch_events":"https://pith.science/api/pith-number/ZU2Y5F6BHIT4UHFAK6L7AZ72VP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZU2Y5F6BHIT4UHFAK6L7AZ72VP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZU2Y5F6BHIT4UHFAK6L7AZ72VP/action/storage_attestation","attest_author":"https://pith.science/pith/ZU2Y5F6BHIT4UHFAK6L7AZ72VP/action/author_attestation","sign_citation":"https://pith.science/pith/ZU2Y5F6BHIT4UHFAK6L7AZ72VP/action/citation_signature","submit_replication":"https://pith.science/pith/ZU2Y5F6BHIT4UHFAK6L7AZ72VP/action/replication_record"}},"created_at":"2026-07-05T11:58:41.653684+00:00","updated_at":"2026-07-05T11:58:41.653684+00:00"}