{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DEEWPRLHGAWIDESX7W45J5CPI7","short_pith_number":"pith:DEEWPRLH","schema_version":"1.0","canonical_sha256":"190967c567302c819257fdb9d4f44f47ce745c241b55d36eee3fbb4336cb09ef","source":{"kind":"arxiv","id":"2504.00294","version":1},"attestation_state":"computed","paper":{"title":"Inference-Time Scaling for Complex Tasks: Where We Stand and What Lies Ahead","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Besmira Nushi, Jingya Chen, John Langford, Lingjiao Chen, Neel Joshi, Safoora Yousefi, Shivam Garg, Vibhav Vineet, Vidhisha Balachandran, Yash Lara, Yue Wu","submitted_at":"2025-03-31T23:40:28Z","abstract_excerpt":"Inference-time scaling can enhance the reasoning capabilities of large language models (LLMs) on complex problems that benefit from step-by-step problem solving. Although lengthening generated scratchpads has proven effective for mathematical tasks, the broader impact of this approach on other tasks remains less clear. In this work, we investigate the benefits and limitations of scaling methods across nine state-of-the-art models and eight challenging tasks, including math and STEM reasoning, calendar planning, NP-hard problems, navigation, and spatial reasoning. We compare conventional models"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.00294","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-31T23:40:28Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"6db01c422ef8733db4f31eba0bffb27f3b92c28ad6fa7e9e485cf94e3f773458","abstract_canon_sha256":"c6d7da67bf2b4acd049bce8c0456dd376b45d3f5a7aa02c22d63fa55c4301a78"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:42:23.907879Z","signature_b64":"nlpObtfq3HPLvfDnCyrAJxQUoLhTcN24pZrB6piWkqhNQmOxp9D5drCSs5FrxWWa4zKX6EZgWF8Z3bOGiqLmDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"190967c567302c819257fdb9d4f44f47ce745c241b55d36eee3fbb4336cb09ef","last_reissued_at":"2026-07-05T10:42:23.907417Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:42:23.907417Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inference-Time Scaling for Complex Tasks: Where We Stand and What Lies Ahead","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Besmira Nushi, Jingya Chen, John Langford, Lingjiao Chen, Neel Joshi, Safoora Yousefi, Shivam Garg, Vibhav Vineet, Vidhisha Balachandran, Yash Lara, Yue Wu","submitted_at":"2025-03-31T23:40:28Z","abstract_excerpt":"Inference-time scaling can enhance the reasoning capabilities of large language models (LLMs) on complex problems that benefit from step-by-step problem solving. Although lengthening generated scratchpads has proven effective for mathematical tasks, the broader impact of this approach on other tasks remains less clear. In this work, we investigate the benefits and limitations of scaling methods across nine state-of-the-art models and eight challenging tasks, including math and STEM reasoning, calendar planning, NP-hard problems, navigation, and spatial reasoning. We compare conventional models"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.00294","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.00294/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.00294","created_at":"2026-07-05T10:42:23.907475+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.00294v1","created_at":"2026-07-05T10:42:23.907475+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.00294","created_at":"2026-07-05T10:42:23.907475+00:00"},{"alias_kind":"pith_short_12","alias_value":"DEEWPRLHGAWI","created_at":"2026-07-05T10:42:23.907475+00:00"},{"alias_kind":"pith_short_16","alias_value":"DEEWPRLHGAWIDESX","created_at":"2026-07-05T10:42:23.907475+00:00"},{"alias_kind":"pith_short_8","alias_value":"DEEWPRLH","created_at":"2026-07-05T10:42:23.907475+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31484","citing_title":"Fork-Think with Confidence","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15134","citing_title":"The Unreasonable Effectiveness of Entropy Minimization in LLM Reasoning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2510.14703","citing_title":"ToolPRM: Fine-Grained Inference Scaling of Structured Outputs for Function Calling","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2504.21318","citing_title":"Phi-4-reasoning Technical Report","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27911","citing_title":"Physical Foundation Models: Fixed hardware implementations of large-scale neural networks","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09519","citing_title":"Weighted Rules under the Stable Model Semantics","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06219","citing_title":"Joint Consistency: A Unified Test-Time Aggregation Framework via Energy Minimization","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DEEWPRLHGAWIDESX7W45J5CPI7","json":"https://pith.science/pith/DEEWPRLHGAWIDESX7W45J5CPI7.json","graph_json":"https://pith.science/api/pith-number/DEEWPRLHGAWIDESX7W45J5CPI7/graph.json","events_json":"https://pith.science/api/pith-number/DEEWPRLHGAWIDESX7W45J5CPI7/events.json","paper":"https://pith.science/paper/DEEWPRLH"},"agent_actions":{"view_html":"https://pith.science/pith/DEEWPRLHGAWIDESX7W45J5CPI7","download_json":"https://pith.science/pith/DEEWPRLHGAWIDESX7W45J5CPI7.json","view_paper":"https://pith.science/paper/DEEWPRLH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.00294&json=true","fetch_graph":"https://pith.science/api/pith-number/DEEWPRLHGAWIDESX7W45J5CPI7/graph.json","fetch_events":"https://pith.science/api/pith-number/DEEWPRLHGAWIDESX7W45J5CPI7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DEEWPRLHGAWIDESX7W45J5CPI7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DEEWPRLHGAWIDESX7W45J5CPI7/action/storage_attestation","attest_author":"https://pith.science/pith/DEEWPRLHGAWIDESX7W45J5CPI7/action/author_attestation","sign_citation":"https://pith.science/pith/DEEWPRLHGAWIDESX7W45J5CPI7/action/citation_signature","submit_replication":"https://pith.science/pith/DEEWPRLHGAWIDESX7W45J5CPI7/action/replication_record"}},"created_at":"2026-07-05T10:42:23.907475+00:00","updated_at":"2026-07-05T10:42:23.907475+00:00"}