{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6XBWKF7W6XIEM4FE4ZUZCIGZQP","short_pith_number":"pith:6XBWKF7W","schema_version":"1.0","canonical_sha256":"f5c36517f6f5d04670a4e6699120d983ca6ce8c08f4dd93e066763dad55f37a8","source":{"kind":"arxiv","id":"2309.15028","version":3},"attestation_state":"computed","paper":{"title":"Don't throw away your value model! Generating more preferable text with Value-Guided Monte-Carlo Tree Search decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Andrew Cohen, Asli Celikyilmaz, Hannaneh Hajishirzi, Jiacheng Liu, Ramakanth Pasunuru, Yejin Choi","submitted_at":"2023-09-26T15:57:57Z","abstract_excerpt":"Inference-time search algorithms such as Monte-Carlo Tree Search (MCTS) may seem unnecessary when generating natural language text based on state-of-the-art reinforcement learning such as Proximal Policy Optimization (PPO). In this paper, we demonstrate that it is possible to get extra mileage out of PPO by integrating MCTS on top. The key idea is not to throw out the value network, a byproduct of PPO training for evaluating partial output sequences, when decoding text out of the policy network. More concretely, we present a novel value-guided decoding algorithm called PPO-MCTS, which can inte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.15028","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-26T15:57:57Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"fa2c3d47e4d98188a681c7941754b968604429b373785ce293c0a1b70179a920","abstract_canon_sha256":"9dec11925715ef63b6ad2021ae34cc615e43424dbc5cc97d38022f37914c3245"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:03:26.717440Z","signature_b64":"GCdRPqp/s3OkLgWGSE93XtsCberhqCkjpaa9NF0s+szfWAzQ7y/FQZrThFs9TJ4E579m53SBsAjgk4vR/i/YDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5c36517f6f5d04670a4e6699120d983ca6ce8c08f4dd93e066763dad55f37a8","last_reissued_at":"2026-07-05T08:03:26.716837Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:03:26.716837Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Don't throw away your value model! Generating more preferable text with Value-Guided Monte-Carlo Tree Search decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Andrew Cohen, Asli Celikyilmaz, Hannaneh Hajishirzi, Jiacheng Liu, Ramakanth Pasunuru, Yejin Choi","submitted_at":"2023-09-26T15:57:57Z","abstract_excerpt":"Inference-time search algorithms such as Monte-Carlo Tree Search (MCTS) may seem unnecessary when generating natural language text based on state-of-the-art reinforcement learning such as Proximal Policy Optimization (PPO). In this paper, we demonstrate that it is possible to get extra mileage out of PPO by integrating MCTS on top. The key idea is not to throw out the value network, a byproduct of PPO training for evaluating partial output sequences, when decoding text out of the policy network. More concretely, we present a novel value-guided decoding algorithm called PPO-MCTS, which can inte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.15028","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.15028/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.15028","created_at":"2026-07-05T08:03:26.716899+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.15028v3","created_at":"2026-07-05T08:03:26.716899+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.15028","created_at":"2026-07-05T08:03:26.716899+00:00"},{"alias_kind":"pith_short_12","alias_value":"6XBWKF7W6XIE","created_at":"2026-07-05T08:03:26.716899+00:00"},{"alias_kind":"pith_short_16","alias_value":"6XBWKF7W6XIEM4FE","created_at":"2026-07-05T08:03:26.716899+00:00"},{"alias_kind":"pith_short_8","alias_value":"6XBWKF7W","created_at":"2026-07-05T08:03:26.716899+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27030","citing_title":"Share More, Search Less: Collaborative Parallel Thinking for Efficient Test-Time Scaling","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2504.12501","citing_title":"Reinforcement Learning from Human Feedback","ref_index":157,"is_internal_anchor":false},{"citing_arxiv_id":"2408.07199","citing_title":"Agent Q: Advanced Reasoning and Learning for Autonomous AI Agents","ref_index":219,"is_internal_anchor":false},{"citing_arxiv_id":"2510.14703","citing_title":"ToolPRM: Fine-Grained Inference Scaling of Structured Outputs for Function Calling","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":152,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03356","citing_title":"POSTCONDBENCH: Benchmarking Correctness and Completeness in Formal Postcondition Inference","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6XBWKF7W6XIEM4FE4ZUZCIGZQP","json":"https://pith.science/pith/6XBWKF7W6XIEM4FE4ZUZCIGZQP.json","graph_json":"https://pith.science/api/pith-number/6XBWKF7W6XIEM4FE4ZUZCIGZQP/graph.json","events_json":"https://pith.science/api/pith-number/6XBWKF7W6XIEM4FE4ZUZCIGZQP/events.json","paper":"https://pith.science/paper/6XBWKF7W"},"agent_actions":{"view_html":"https://pith.science/pith/6XBWKF7W6XIEM4FE4ZUZCIGZQP","download_json":"https://pith.science/pith/6XBWKF7W6XIEM4FE4ZUZCIGZQP.json","view_paper":"https://pith.science/paper/6XBWKF7W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.15028&json=true","fetch_graph":"https://pith.science/api/pith-number/6XBWKF7W6XIEM4FE4ZUZCIGZQP/graph.json","fetch_events":"https://pith.science/api/pith-number/6XBWKF7W6XIEM4FE4ZUZCIGZQP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6XBWKF7W6XIEM4FE4ZUZCIGZQP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6XBWKF7W6XIEM4FE4ZUZCIGZQP/action/storage_attestation","attest_author":"https://pith.science/pith/6XBWKF7W6XIEM4FE4ZUZCIGZQP/action/author_attestation","sign_citation":"https://pith.science/pith/6XBWKF7W6XIEM4FE4ZUZCIGZQP/action/citation_signature","submit_replication":"https://pith.science/pith/6XBWKF7W6XIEM4FE4ZUZCIGZQP/action/replication_record"}},"created_at":"2026-07-05T08:03:26.716899+00:00","updated_at":"2026-07-05T08:03:26.716899+00:00"}