{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PJ2ZM3SBT7GJU7AADIEUFB77WE","short_pith_number":"pith:PJ2ZM3SB","schema_version":"1.0","canonical_sha256":"7a75966e419fcc9a7c001a094287ffb10e564f99fdf4e124fa473ba14dd42899","source":{"kind":"arxiv","id":"2502.12521","version":1},"attestation_state":"computed","paper":{"title":"Inference-Time Computations for LLM Reasoning and Planning: A Benchmark and Insights","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Blake Olson, Eric Li, Hongyi Ling, James Caverlee, Sambhav Khurana, Shubham Parashar, Shuiwang Ji","submitted_at":"2025-02-18T04:11:29Z","abstract_excerpt":"We examine the reasoning and planning capabilities of large language models (LLMs) in solving complex tasks. Recent advances in inference-time techniques demonstrate the potential to enhance LLM reasoning without additional training by exploring intermediate steps during inference. Notably, OpenAI's o1 model shows promising performance through its novel use of multi-step reasoning and verification. Here, we explore how scaling inference-time techniques can improve reasoning and planning, focusing on understanding the tradeoff between computational cost and performance. To this end, we construc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.12521","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-02-18T04:11:29Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7d2464756ebdd2433eeffcaf4e76bd530d45737adf1d34f7a777dce3bb78d62b","abstract_canon_sha256":"08c1c7266286402e7c297ee9353950a5def116e5edfda345d3c49bde14dc0eec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:56.897333Z","signature_b64":"P4h0FJexCXhyUiKKjQ7d7JrWwrSGAAmDch0g5e7241mOv9OGYNQhYhpVhYdlnNjKa6uP37AWH8rsIRGLw1vzCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7a75966e419fcc9a7c001a094287ffb10e564f99fdf4e124fa473ba14dd42899","last_reissued_at":"2026-07-05T10:15:56.896813Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:56.896813Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inference-Time Computations for LLM Reasoning and Planning: A Benchmark and Insights","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Blake Olson, Eric Li, Hongyi Ling, James Caverlee, Sambhav Khurana, Shubham Parashar, Shuiwang Ji","submitted_at":"2025-02-18T04:11:29Z","abstract_excerpt":"We examine the reasoning and planning capabilities of large language models (LLMs) in solving complex tasks. Recent advances in inference-time techniques demonstrate the potential to enhance LLM reasoning without additional training by exploring intermediate steps during inference. Notably, OpenAI's o1 model shows promising performance through its novel use of multi-step reasoning and verification. Here, we explore how scaling inference-time techniques can improve reasoning and planning, focusing on understanding the tradeoff between computational cost and performance. To this end, we construc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.12521","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.12521/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.12521","created_at":"2026-07-05T10:15:56.896883+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.12521v1","created_at":"2026-07-05T10:15:56.896883+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.12521","created_at":"2026-07-05T10:15:56.896883+00:00"},{"alias_kind":"pith_short_12","alias_value":"PJ2ZM3SBT7GJ","created_at":"2026-07-05T10:15:56.896883+00:00"},{"alias_kind":"pith_short_16","alias_value":"PJ2ZM3SBT7GJU7AA","created_at":"2026-07-05T10:15:56.896883+00:00"},{"alias_kind":"pith_short_8","alias_value":"PJ2ZM3SB","created_at":"2026-07-05T10:15:56.896883+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08596","citing_title":"Distilling LLM Reasoning into an Interpretable Policy Tree for Human-AI Collaboration","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28277","citing_title":"Do LLMs Build World Models From Text? A Multilingual Diagnostic of Spatial Reasoning","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04000","citing_title":"Mitigating False Positives in Static Memory Safety Analysis of Rust Programs via Reinforcement Learning","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08057","citing_title":"CA-SQL: Complexity-Aware Inference Time Reasoning for Text-to-SQL via Exploration and Compute Budget Allocation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16913","citing_title":"The Cognitive Penalty: Ablating System 1 and System 2 Reasoning in Edge-Native SLMs for Decentralized Consensus","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04000","citing_title":"Mitigating False Positives in Static Memory Safety Analysis of Rust Programs via Reinforcement Learning","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PJ2ZM3SBT7GJU7AADIEUFB77WE","json":"https://pith.science/pith/PJ2ZM3SBT7GJU7AADIEUFB77WE.json","graph_json":"https://pith.science/api/pith-number/PJ2ZM3SBT7GJU7AADIEUFB77WE/graph.json","events_json":"https://pith.science/api/pith-number/PJ2ZM3SBT7GJU7AADIEUFB77WE/events.json","paper":"https://pith.science/paper/PJ2ZM3SB"},"agent_actions":{"view_html":"https://pith.science/pith/PJ2ZM3SBT7GJU7AADIEUFB77WE","download_json":"https://pith.science/pith/PJ2ZM3SBT7GJU7AADIEUFB77WE.json","view_paper":"https://pith.science/paper/PJ2ZM3SB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.12521&json=true","fetch_graph":"https://pith.science/api/pith-number/PJ2ZM3SBT7GJU7AADIEUFB77WE/graph.json","fetch_events":"https://pith.science/api/pith-number/PJ2ZM3SBT7GJU7AADIEUFB77WE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PJ2ZM3SBT7GJU7AADIEUFB77WE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PJ2ZM3SBT7GJU7AADIEUFB77WE/action/storage_attestation","attest_author":"https://pith.science/pith/PJ2ZM3SBT7GJU7AADIEUFB77WE/action/author_attestation","sign_citation":"https://pith.science/pith/PJ2ZM3SBT7GJU7AADIEUFB77WE/action/citation_signature","submit_replication":"https://pith.science/pith/PJ2ZM3SBT7GJU7AADIEUFB77WE/action/replication_record"}},"created_at":"2026-07-05T10:15:56.896883+00:00","updated_at":"2026-07-05T10:15:56.896883+00:00"}