{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:I7WRJX3TKR6MELU3O7FQQU2OGC","short_pith_number":"pith:I7WRJX3T","schema_version":"1.0","canonical_sha256":"47ed14df73547cc22e9b77cb08534e30982505693819442a2c5cccf37ef49fca","source":{"kind":"arxiv","id":"2503.00031","version":1},"attestation_state":"computed","paper":{"title":"Efficient Test-Time Scaling via Self-Calibration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chengsong Huang, Jiacheng Liu, Jiaxin Huang, Jixuan Leng, Langlin Huang","submitted_at":"2025-02-25T00:21:14Z","abstract_excerpt":"Increasing test-time computation is a straightforward approach to enhancing the quality of responses in Large Language Models (LLMs). While Best-of-N sampling and Self-Consistency with majority voting are simple and effective, they require a fixed number of sampling responses for each query, regardless of its complexity. This could result in wasted computation for simpler questions and insufficient exploration for more challenging ones. In this work, we argue that model confidence of responses can be used for improving the efficiency of test-time scaling. Unfortunately, LLMs are known to be ov"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.00031","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-25T00:21:14Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a2dab5064cb1ed8405b6ba8d725912969befa09fb8e332965cf0f81f9e9cade1","abstract_canon_sha256":"f80612564d54acd1a8a23f09dda88832bdeb6df6931bc1fcec2040c076c1871f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:16.393493Z","signature_b64":"smaxK1nPfmObC3Jm9rZOgrkRy0mklCTIaSK9Y3J3/rMn3KxoHE98NyLUwWq9B5XHch3ibvJWARH7Fal7CKNxCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"47ed14df73547cc22e9b77cb08534e30982505693819442a2c5cccf37ef49fca","last_reissued_at":"2026-07-05T10:22:16.392954Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:16.392954Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Test-Time Scaling via Self-Calibration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chengsong Huang, Jiacheng Liu, Jiaxin Huang, Jixuan Leng, Langlin Huang","submitted_at":"2025-02-25T00:21:14Z","abstract_excerpt":"Increasing test-time computation is a straightforward approach to enhancing the quality of responses in Large Language Models (LLMs). While Best-of-N sampling and Self-Consistency with majority voting are simple and effective, they require a fixed number of sampling responses for each query, regardless of its complexity. This could result in wasted computation for simpler questions and insufficient exploration for more challenging ones. In this work, we argue that model confidence of responses can be used for improving the efficiency of test-time scaling. Unfortunately, LLMs are known to be ov"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.00031","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.00031/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.00031","created_at":"2026-07-05T10:22:16.393019+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.00031v1","created_at":"2026-07-05T10:22:16.393019+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.00031","created_at":"2026-07-05T10:22:16.393019+00:00"},{"alias_kind":"pith_short_12","alias_value":"I7WRJX3TKR6M","created_at":"2026-07-05T10:22:16.393019+00:00"},{"alias_kind":"pith_short_16","alias_value":"I7WRJX3TKR6MELU3","created_at":"2026-07-05T10:22:16.393019+00:00"},{"alias_kind":"pith_short_8","alias_value":"I7WRJX3T","created_at":"2026-07-05T10:22:16.393019+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12935","citing_title":"MARS: Margin-Adversarial Risk-controlled Stopping for Parallel LLM Test-time Scaling","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03102","citing_title":"Small RL Controller, Large Language Model: RL-Guided Adaptive Sampling for Test-Time Scaling","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15529","citing_title":"Process Rewards with Learned Reliability","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09438","citing_title":"GrACE: A Generative Approach to Better Confidence Elicitation and Efficient Test-Time Scaling in Large Language Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16419","citing_title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08083","citing_title":"LLMs Improving LLMs: Agentic Discovery for Test-Time Scaling","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":283,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23333","citing_title":"Process Supervision of Confidence Margin for Calibrated LLM Reasoning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06165","citing_title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","ref_index":179,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05566","citing_title":"Nonsense Helps: Prompt Space Perturbation Broadens Reasoning Exploration","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01194","citing_title":"VLA-ATTC: Adaptive Test-Time Compute for VLA Models with Relative Action Critic Model","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02290","citing_title":"Distilling Long-CoT Reasoning through Collaborative Step-wise Multi-Teacher Decoding","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08083","citing_title":"LLMs Improving LLMs: Agentic Discovery for Test-Time Scaling","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I7WRJX3TKR6MELU3O7FQQU2OGC","json":"https://pith.science/pith/I7WRJX3TKR6MELU3O7FQQU2OGC.json","graph_json":"https://pith.science/api/pith-number/I7WRJX3TKR6MELU3O7FQQU2OGC/graph.json","events_json":"https://pith.science/api/pith-number/I7WRJX3TKR6MELU3O7FQQU2OGC/events.json","paper":"https://pith.science/paper/I7WRJX3T"},"agent_actions":{"view_html":"https://pith.science/pith/I7WRJX3TKR6MELU3O7FQQU2OGC","download_json":"https://pith.science/pith/I7WRJX3TKR6MELU3O7FQQU2OGC.json","view_paper":"https://pith.science/paper/I7WRJX3T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.00031&json=true","fetch_graph":"https://pith.science/api/pith-number/I7WRJX3TKR6MELU3O7FQQU2OGC/graph.json","fetch_events":"https://pith.science/api/pith-number/I7WRJX3TKR6MELU3O7FQQU2OGC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I7WRJX3TKR6MELU3O7FQQU2OGC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I7WRJX3TKR6MELU3O7FQQU2OGC/action/storage_attestation","attest_author":"https://pith.science/pith/I7WRJX3TKR6MELU3O7FQQU2OGC/action/author_attestation","sign_citation":"https://pith.science/pith/I7WRJX3TKR6MELU3O7FQQU2OGC/action/citation_signature","submit_replication":"https://pith.science/pith/I7WRJX3TKR6MELU3O7FQQU2OGC/action/replication_record"}},"created_at":"2026-07-05T10:22:16.393019+00:00","updated_at":"2026-07-05T10:22:16.393019+00:00"}