{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7CRPBYN7V7PTC5ZTXVPNHRVJCS","short_pith_number":"pith:7CRPBYN7","schema_version":"1.0","canonical_sha256":"f8a2f0e1bfafdf317733bd5ed3c6a914b7e8405781ded5aacb89477972904761","source":{"kind":"arxiv","id":"2506.12928","version":1},"attestation_state":"computed","paper":{"title":"Scaling Test-time Compute for LLM Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Changwang Zhang, Chenghua Lin, Dehua Ma, Ge Zhang, Hanhao Li, Jiaheng Liu, Jian Yang, Jun Wang, King Zhu, Minghao Liu, Siwei Wu, Tianshun Xing, Wangchunshu Zhou, Xiangru Tang, Yuchen Eleanor Jiang","submitted_at":"2025-06-15T17:59:47Z","abstract_excerpt":"Scaling test time compute has shown remarkable success in improving the reasoning abilities of large language models (LLMs). In this work, we conduct the first systematic exploration of applying test-time scaling methods to language agents and investigate the extent to which it improves their effectiveness. Specifically, we explore different test-time scaling strategies, including: (1) parallel sampling algorithms; (2) sequential revision strategies; (3) verifiers and merging methods; (4)strategies for diversifying rollouts.We carefully analyze and ablate the impact of different design strateg"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.12928","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-15T17:59:47Z","cross_cats_sorted":[],"title_canon_sha256":"357a3369651c905ad5f84be4cd060aca14443a587c6982ae41ab0e9e58f0b783","abstract_canon_sha256":"cf6b8efcd864faf1932a3140c2afaf9605b2473f9748a2dd77de43d3216f7e0a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:02.526389Z","signature_b64":"IEFR5Z3jY8iStyNJrLIDyjvCm03yZEmRJr3tqS542UM5EhS+3PmUYDyX43hQxRXQY3zzQIpJrNuBMLaD8bidCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f8a2f0e1bfafdf317733bd5ed3c6a914b7e8405781ded5aacb89477972904761","last_reissued_at":"2026-07-05T11:22:02.525985Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:02.525985Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Test-time Compute for LLM Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Changwang Zhang, Chenghua Lin, Dehua Ma, Ge Zhang, Hanhao Li, Jiaheng Liu, Jian Yang, Jun Wang, King Zhu, Minghao Liu, Siwei Wu, Tianshun Xing, Wangchunshu Zhou, Xiangru Tang, Yuchen Eleanor Jiang","submitted_at":"2025-06-15T17:59:47Z","abstract_excerpt":"Scaling test time compute has shown remarkable success in improving the reasoning abilities of large language models (LLMs). In this work, we conduct the first systematic exploration of applying test-time scaling methods to language agents and investigate the extent to which it improves their effectiveness. Specifically, we explore different test-time scaling strategies, including: (1) parallel sampling algorithms; (2) sequential revision strategies; (3) verifiers and merging methods; (4)strategies for diversifying rollouts.We carefully analyze and ablate the impact of different design strateg"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.12928","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.12928/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.12928","created_at":"2026-07-05T11:22:02.526048+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.12928v1","created_at":"2026-07-05T11:22:02.526048+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.12928","created_at":"2026-07-05T11:22:02.526048+00:00"},{"alias_kind":"pith_short_12","alias_value":"7CRPBYN7V7PT","created_at":"2026-07-05T11:22:02.526048+00:00"},{"alias_kind":"pith_short_16","alias_value":"7CRPBYN7V7PTC5ZT","created_at":"2026-07-05T11:22:02.526048+00:00"},{"alias_kind":"pith_short_8","alias_value":"7CRPBYN7","created_at":"2026-07-05T11:22:02.526048+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22154","citing_title":"IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22102","citing_title":"ExComm: Exploration-Stage Communication for Error-Resilient Agentic Test-Time Scaling","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17320","citing_title":"TClone: Low-Latency Forking of Live GUI Environments for Computer-Use Agents","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14037","citing_title":"Self-Pruned Key-Value Attention: Learning When to Write by Predicting Future Utility","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11039","citing_title":"The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09121","citing_title":"A Communication-Theoretic Framework for LLM Agents: Cost-Aware Adaptive Reliability","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05701","citing_title":"Inference-Time Budget Control for LLM Search Agents","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19341","citing_title":"Evaluation-driven Scaling for Scientific Discovery","ref_index":176,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7CRPBYN7V7PTC5ZTXVPNHRVJCS","json":"https://pith.science/pith/7CRPBYN7V7PTC5ZTXVPNHRVJCS.json","graph_json":"https://pith.science/api/pith-number/7CRPBYN7V7PTC5ZTXVPNHRVJCS/graph.json","events_json":"https://pith.science/api/pith-number/7CRPBYN7V7PTC5ZTXVPNHRVJCS/events.json","paper":"https://pith.science/paper/7CRPBYN7"},"agent_actions":{"view_html":"https://pith.science/pith/7CRPBYN7V7PTC5ZTXVPNHRVJCS","download_json":"https://pith.science/pith/7CRPBYN7V7PTC5ZTXVPNHRVJCS.json","view_paper":"https://pith.science/paper/7CRPBYN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.12928&json=true","fetch_graph":"https://pith.science/api/pith-number/7CRPBYN7V7PTC5ZTXVPNHRVJCS/graph.json","fetch_events":"https://pith.science/api/pith-number/7CRPBYN7V7PTC5ZTXVPNHRVJCS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7CRPBYN7V7PTC5ZTXVPNHRVJCS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7CRPBYN7V7PTC5ZTXVPNHRVJCS/action/storage_attestation","attest_author":"https://pith.science/pith/7CRPBYN7V7PTC5ZTXVPNHRVJCS/action/author_attestation","sign_citation":"https://pith.science/pith/7CRPBYN7V7PTC5ZTXVPNHRVJCS/action/citation_signature","submit_replication":"https://pith.science/pith/7CRPBYN7V7PTC5ZTXVPNHRVJCS/action/replication_record"}},"created_at":"2026-07-05T11:22:02.526048+00:00","updated_at":"2026-07-05T11:22:02.526048+00:00"}