{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HEPGQWXMWEEBCVD47ZQ4K4RGNE","short_pith_number":"pith:HEPGQWXM","schema_version":"1.0","canonical_sha256":"391e685aecb10811547cfe61c57226691be82343c56bef72e7b8a698d7bae1de","source":{"kind":"arxiv","id":"2402.05201","version":3},"attestation_state":"computed","paper":{"title":"The Effect of Sampling Temperature on Problem Solving in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Erhan Guven, Matthew Renze","submitted_at":"2024-02-07T19:11:23Z","abstract_excerpt":"In this research study, we empirically investigate the effect of sampling temperature on the performance of Large Language Models (LLMs) on various problem-solving tasks. We created a multiple-choice question-and-answer (MCQA) exam by randomly sampling problems from standard LLM benchmarks. Then, we used nine popular LLMs with five prompt-engineering techniques to solve the MCQA problems while increasing the sampling temperature from 0.0 to 1.6. Despite anecdotal reports to the contrary, our empirical results indicate that changes in temperature from 0.0 to 1.0 do not have a statistically sign"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05201","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-07T19:11:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"561f001d8ccb7bc095e2caaeab0a8fe85317e7087b4b83b03d685d819ea691be","abstract_canon_sha256":"927a4cbd5bb7ef5e9e086388c6d7021bbec1e27400c33471caf3a8ba7de1f39c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:30:52.792273Z","signature_b64":"e3EZzbcMe/h78pwOPi4wPBuTw/anJRdRekHWMMk+wrTsvktCwG6CWeuMTCXyLH9k7OG5A7suljVJgJ5lJrlUBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"391e685aecb10811547cfe61c57226691be82343c56bef72e7b8a698d7bae1de","last_reissued_at":"2026-07-05T10:30:52.791487Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:30:52.791487Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Effect of Sampling Temperature on Problem Solving in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Erhan Guven, Matthew Renze","submitted_at":"2024-02-07T19:11:23Z","abstract_excerpt":"In this research study, we empirically investigate the effect of sampling temperature on the performance of Large Language Models (LLMs) on various problem-solving tasks. We created a multiple-choice question-and-answer (MCQA) exam by randomly sampling problems from standard LLM benchmarks. Then, we used nine popular LLMs with five prompt-engineering techniques to solve the MCQA problems while increasing the sampling temperature from 0.0 to 1.6. Despite anecdotal reports to the contrary, our empirical results indicate that changes in temperature from 0.0 to 1.0 do not have a statistically sign"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05201","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05201/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05201","created_at":"2026-07-05T10:30:52.791581+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05201v3","created_at":"2026-07-05T10:30:52.791581+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05201","created_at":"2026-07-05T10:30:52.791581+00:00"},{"alias_kind":"pith_short_12","alias_value":"HEPGQWXMWEEB","created_at":"2026-07-05T10:30:52.791581+00:00"},{"alias_kind":"pith_short_16","alias_value":"HEPGQWXMWEEBCVD4","created_at":"2026-07-05T10:30:52.791581+00:00"},{"alias_kind":"pith_short_8","alias_value":"HEPGQWXM","created_at":"2026-07-05T10:30:52.791581+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21228","citing_title":"Sakana Fugu Technical Report","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2503.17181","citing_title":"A Study of LLMs' Preferences for Libraries and Programming Languages","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12858","citing_title":"Information-Consistent Language Model Recommendations through Group Relative Policy Optimization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19349","citing_title":"ShinkaEvolve: Towards Open-Ended And Sample-Efficient Program Evolution","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04066","citing_title":"Adapt to Thrive! Adaptive Power-Mean Policy Optimization for Improved LLM Reasoning","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04065","citing_title":"Free Energy-Driven Reinforcement Learning with Adaptive Advantage Shaping for Unsupervised Reasoning in LLMs","ref_index":131,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HEPGQWXMWEEBCVD47ZQ4K4RGNE","json":"https://pith.science/pith/HEPGQWXMWEEBCVD47ZQ4K4RGNE.json","graph_json":"https://pith.science/api/pith-number/HEPGQWXMWEEBCVD47ZQ4K4RGNE/graph.json","events_json":"https://pith.science/api/pith-number/HEPGQWXMWEEBCVD47ZQ4K4RGNE/events.json","paper":"https://pith.science/paper/HEPGQWXM"},"agent_actions":{"view_html":"https://pith.science/pith/HEPGQWXMWEEBCVD47ZQ4K4RGNE","download_json":"https://pith.science/pith/HEPGQWXMWEEBCVD47ZQ4K4RGNE.json","view_paper":"https://pith.science/paper/HEPGQWXM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05201&json=true","fetch_graph":"https://pith.science/api/pith-number/HEPGQWXMWEEBCVD47ZQ4K4RGNE/graph.json","fetch_events":"https://pith.science/api/pith-number/HEPGQWXMWEEBCVD47ZQ4K4RGNE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HEPGQWXMWEEBCVD47ZQ4K4RGNE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HEPGQWXMWEEBCVD47ZQ4K4RGNE/action/storage_attestation","attest_author":"https://pith.science/pith/HEPGQWXMWEEBCVD47ZQ4K4RGNE/action/author_attestation","sign_citation":"https://pith.science/pith/HEPGQWXMWEEBCVD47ZQ4K4RGNE/action/citation_signature","submit_replication":"https://pith.science/pith/HEPGQWXMWEEBCVD47ZQ4K4RGNE/action/replication_record"}},"created_at":"2026-07-05T10:30:52.791581+00:00","updated_at":"2026-07-05T10:30:52.791581+00:00"}