{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Z6COCQ6H2KSHDWMPYIBGQJGQA7","short_pith_number":"pith:Z6COCQ6H","schema_version":"1.0","canonical_sha256":"cf84e143c7d2a471d98fc2026824d007c0e52977749d2a6485be3d4aafda2b54","source":{"kind":"arxiv","id":"2411.11694","version":4},"attestation_state":"computed","paper":{"title":"Enhancing LLM Reasoning with Reward-guided Tree Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dong Yan, Haoxiang Sun, Jia Deng, Jian Xie, Jiapeng Wang, Jie Chen, Jinhao Jiang, Ji-Rong Wen, Wayne Xin Zhao, Xiaoxue Cheng, Yingqian Min, Yiru Tang, Zheng Liu, Zhipeng Chen, Zhongyuan Wang","submitted_at":"2024-11-18T16:15:17Z","abstract_excerpt":"Recently, test-time scaling has garnered significant attention from the research community, largely due to the substantial advancements of the o1 model released by OpenAI. By allocating more computational resources during the inference phase, large language models~(LLMs) can extensively explore the solution space by generating more thought tokens or diverse solutions, thereby producing more accurate responses. However, developing an o1-like reasoning approach is challenging, and researchers have been making various attempts to advance this open area of research. In this paper, we present a pre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.11694","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-18T16:15:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3dbe3d519ec2c5006aecb45504f71e92a0c53faf120e4a145799b6af8b931501","abstract_canon_sha256":"5fd91fc21c21d554b3afb28afe6dec67968f224f714d010a0c2cbe86c4364bcf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:35.456433Z","signature_b64":"xDgIjK1c/osZ1bv0p9j/mgrKzBCs7hbu0ByOG6iZtsxUhLrnzDyJy7b9JDDkroR+pC3C7OxV16xvNB+aduFVCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf84e143c7d2a471d98fc2026824d007c0e52977749d2a6485be3d4aafda2b54","last_reissued_at":"2026-07-05T09:55:35.455874Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:35.455874Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing LLM Reasoning with Reward-guided Tree Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dong Yan, Haoxiang Sun, Jia Deng, Jian Xie, Jiapeng Wang, Jie Chen, Jinhao Jiang, Ji-Rong Wen, Wayne Xin Zhao, Xiaoxue Cheng, Yingqian Min, Yiru Tang, Zheng Liu, Zhipeng Chen, Zhongyuan Wang","submitted_at":"2024-11-18T16:15:17Z","abstract_excerpt":"Recently, test-time scaling has garnered significant attention from the research community, largely due to the substantial advancements of the o1 model released by OpenAI. By allocating more computational resources during the inference phase, large language models~(LLMs) can extensively explore the solution space by generating more thought tokens or diverse solutions, thereby producing more accurate responses. However, developing an o1-like reasoning approach is challenging, and researchers have been making various attempts to advance this open area of research. In this paper, we present a pre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.11694","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.11694/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.11694","created_at":"2026-07-05T09:55:35.455931+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.11694v4","created_at":"2026-07-05T09:55:35.455931+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.11694","created_at":"2026-07-05T09:55:35.455931+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z6COCQ6H2KSH","created_at":"2026-07-05T09:55:35.455931+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z6COCQ6H2KSHDWMP","created_at":"2026-07-05T09:55:35.455931+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z6COCQ6H","created_at":"2026-07-05T09:55:35.455931+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":285,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17449","citing_title":"MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":284,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10932","citing_title":"Density Field State Space Models: 1-Bit Distillation, Efficient Inference, and Knowledge Organization in Mamba-2","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20599","citing_title":"Beyond Fixed Budgets: Characterizing the Inelasticity and Limitations of Tree-of-Thought Reasoning Strategies","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21460","citing_title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2505.04588","citing_title":"ZeroSearch: Incentivize the Search Capability of LLMs without Searching","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2602.03433","citing_title":"When control meets large language models: From words to dynamics","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21743","citing_title":"Retrieval-of-Thought: Efficient Reasoning via Reusing Thoughts","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2412.09413","citing_title":"Imitate, Explore, and Self-Improve: A Reproduction Report on Slow-thinking Reasoning Systems","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2505.04588","citing_title":"ZeroSearch: Incentivize the Search Capability of LLMs without Searching","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":228,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05366","citing_title":"Search-o1: Agentic Search-Enhanced Large Reasoning Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09121","citing_title":"A Communication-Theoretic Framework for LLM Agents: Cost-Aware Adaptive Reliability","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18327","citing_title":"PARM: Pipeline-Adapted Reward Model","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z6COCQ6H2KSHDWMPYIBGQJGQA7","json":"https://pith.science/pith/Z6COCQ6H2KSHDWMPYIBGQJGQA7.json","graph_json":"https://pith.science/api/pith-number/Z6COCQ6H2KSHDWMPYIBGQJGQA7/graph.json","events_json":"https://pith.science/api/pith-number/Z6COCQ6H2KSHDWMPYIBGQJGQA7/events.json","paper":"https://pith.science/paper/Z6COCQ6H"},"agent_actions":{"view_html":"https://pith.science/pith/Z6COCQ6H2KSHDWMPYIBGQJGQA7","download_json":"https://pith.science/pith/Z6COCQ6H2KSHDWMPYIBGQJGQA7.json","view_paper":"https://pith.science/paper/Z6COCQ6H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.11694&json=true","fetch_graph":"https://pith.science/api/pith-number/Z6COCQ6H2KSHDWMPYIBGQJGQA7/graph.json","fetch_events":"https://pith.science/api/pith-number/Z6COCQ6H2KSHDWMPYIBGQJGQA7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z6COCQ6H2KSHDWMPYIBGQJGQA7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z6COCQ6H2KSHDWMPYIBGQJGQA7/action/storage_attestation","attest_author":"https://pith.science/pith/Z6COCQ6H2KSHDWMPYIBGQJGQA7/action/author_attestation","sign_citation":"https://pith.science/pith/Z6COCQ6H2KSHDWMPYIBGQJGQA7/action/citation_signature","submit_replication":"https://pith.science/pith/Z6COCQ6H2KSHDWMPYIBGQJGQA7/action/replication_record"}},"created_at":"2026-07-05T09:55:35.455931+00:00","updated_at":"2026-07-05T09:55:35.455931+00:00"}