{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WPQA7PD3GRUT4CZR2YEKRCXIOW","short_pith_number":"pith:WPQA7PD3","schema_version":"1.0","canonical_sha256":"b3e00fbc7b34693e0b31d608a88ae8759e5391e7753bd4beb533819a4f29a200","source":{"kind":"arxiv","id":"2407.05778","version":1},"attestation_state":"computed","paper":{"title":"When is the consistent prediction likely to be a correct prediction?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alex Nguyen, Chengyu Dong, Dheeraj Mekala, Jingbo Shang","submitted_at":"2024-07-08T09:37:27Z","abstract_excerpt":"Self-consistency (Wang et al., 2023) suggests that the most consistent answer obtained through large language models (LLMs) is more likely to be correct. In this paper, we challenge this argument and propose a nuanced correction. Our observations indicate that consistent answers derived through more computation i.e. longer reasoning texts, rather than simply the most consistent answer across all outputs, are more likely to be correct. This is predominantly because we demonstrate that LLMs can autonomously produce chain-of-thought (CoT) style reasoning with no custom prompts merely while genera"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.05778","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-08T09:37:27Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2e65a56249b15e97ceb0dbebf4a1471cbbb6be017d5e0ffd1e828f4e5b96c1c4","abstract_canon_sha256":"c894d85d761b23e51201a46c2d151cf12d232a4bbefb532358ab0ed5fd0b24a5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:18.590253Z","signature_b64":"dWN9YVee6otytI29F+J7cHrPRbtKeZLKIGQhvLBzaIhq5qamPCP8BtUP1KsUQ0Kst9R7NJm2CAf/9GisqPxoAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b3e00fbc7b34693e0b31d608a88ae8759e5391e7753bd4beb533819a4f29a200","last_reissued_at":"2026-07-05T08:41:18.589753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:18.589753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When is the consistent prediction likely to be a correct prediction?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alex Nguyen, Chengyu Dong, Dheeraj Mekala, Jingbo Shang","submitted_at":"2024-07-08T09:37:27Z","abstract_excerpt":"Self-consistency (Wang et al., 2023) suggests that the most consistent answer obtained through large language models (LLMs) is more likely to be correct. In this paper, we challenge this argument and propose a nuanced correction. Our observations indicate that consistent answers derived through more computation i.e. longer reasoning texts, rather than simply the most consistent answer across all outputs, are more likely to be correct. This is predominantly because we demonstrate that LLMs can autonomously produce chain-of-thought (CoT) style reasoning with no custom prompts merely while genera"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.05778","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.05778/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.05778","created_at":"2026-07-05T08:41:18.589805+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.05778v1","created_at":"2026-07-05T08:41:18.589805+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.05778","created_at":"2026-07-05T08:41:18.589805+00:00"},{"alias_kind":"pith_short_12","alias_value":"WPQA7PD3GRUT","created_at":"2026-07-05T08:41:18.589805+00:00"},{"alias_kind":"pith_short_16","alias_value":"WPQA7PD3GRUT4CZR","created_at":"2026-07-05T08:41:18.589805+00:00"},{"alias_kind":"pith_short_8","alias_value":"WPQA7PD3","created_at":"2026-07-05T08:41:18.589805+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25143","citing_title":"Beyond the Frontier: Stochastic Backtracking for Efficient Test-Time Scaling","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2407.21787","citing_title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WPQA7PD3GRUT4CZR2YEKRCXIOW","json":"https://pith.science/pith/WPQA7PD3GRUT4CZR2YEKRCXIOW.json","graph_json":"https://pith.science/api/pith-number/WPQA7PD3GRUT4CZR2YEKRCXIOW/graph.json","events_json":"https://pith.science/api/pith-number/WPQA7PD3GRUT4CZR2YEKRCXIOW/events.json","paper":"https://pith.science/paper/WPQA7PD3"},"agent_actions":{"view_html":"https://pith.science/pith/WPQA7PD3GRUT4CZR2YEKRCXIOW","download_json":"https://pith.science/pith/WPQA7PD3GRUT4CZR2YEKRCXIOW.json","view_paper":"https://pith.science/paper/WPQA7PD3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.05778&json=true","fetch_graph":"https://pith.science/api/pith-number/WPQA7PD3GRUT4CZR2YEKRCXIOW/graph.json","fetch_events":"https://pith.science/api/pith-number/WPQA7PD3GRUT4CZR2YEKRCXIOW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WPQA7PD3GRUT4CZR2YEKRCXIOW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WPQA7PD3GRUT4CZR2YEKRCXIOW/action/storage_attestation","attest_author":"https://pith.science/pith/WPQA7PD3GRUT4CZR2YEKRCXIOW/action/author_attestation","sign_citation":"https://pith.science/pith/WPQA7PD3GRUT4CZR2YEKRCXIOW/action/citation_signature","submit_replication":"https://pith.science/pith/WPQA7PD3GRUT4CZR2YEKRCXIOW/action/replication_record"}},"created_at":"2026-07-05T08:41:18.589805+00:00","updated_at":"2026-07-05T08:41:18.589805+00:00"}