{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5KT5NQYAVFG2UCD3DGDTAXOJ4C","short_pith_number":"pith:5KT5NQYA","schema_version":"1.0","canonical_sha256":"eaa7d6c300a94daa087b1987305dc9e0bed99868b6cd793fe162d11e4819f6d6","source":{"kind":"arxiv","id":"2402.18563","version":1},"attestation_state":"computed","paper":{"title":"Approaching Human-Level Forecasting with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.IR"],"primary_cat":"cs.LG","authors_text":"Chen Yueh-Han, Danny Halawi, Fred Zhang, Jacob Steinhardt","submitted_at":"2024-02-28T18:54:18Z","abstract_excerpt":"Forecasting future events is important for policy and decision making. In this work, we study whether language models (LMs) can forecast at the level of competitive human forecasters. Towards this goal, we develop a retrieval-augmented LM system designed to automatically search for relevant information, generate forecasts, and aggregate predictions. To facilitate our study, we collect a large dataset of questions from competitive forecasting platforms. Under a test set published after the knowledge cut-offs of our LMs, we evaluate the end-to-end performance of our system against the aggregates"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.18563","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-28T18:54:18Z","cross_cats_sorted":["cs.AI","cs.CL","cs.IR"],"title_canon_sha256":"4a9c96b4eac217da04797debba7f10cd9835ddd244ddcb63aa2ee4da4065e037","abstract_canon_sha256":"ba9e2e17aee463a20004b3568f82f9520683e0428336697d305750a524648d12"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:50:17.370584Z","signature_b64":"LNDblyENTAARw2oJAKPhxoyHI4BPBYYyUmS+yWEUMspzX/qhtDndn6bn9KKWW3H2T69gM8yjMtRTmucdK8AiDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eaa7d6c300a94daa087b1987305dc9e0bed99868b6cd793fe162d11e4819f6d6","last_reissued_at":"2026-07-05T07:50:17.370028Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:50:17.370028Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Approaching Human-Level Forecasting with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.IR"],"primary_cat":"cs.LG","authors_text":"Chen Yueh-Han, Danny Halawi, Fred Zhang, Jacob Steinhardt","submitted_at":"2024-02-28T18:54:18Z","abstract_excerpt":"Forecasting future events is important for policy and decision making. In this work, we study whether language models (LMs) can forecast at the level of competitive human forecasters. Towards this goal, we develop a retrieval-augmented LM system designed to automatically search for relevant information, generate forecasts, and aggregate predictions. To facilitate our study, we collect a large dataset of questions from competitive forecasting platforms. Under a test set published after the knowledge cut-offs of our LMs, we evaluate the end-to-end performance of our system against the aggregates"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.18563","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.18563/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.18563","created_at":"2026-07-05T07:50:17.370097+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.18563v1","created_at":"2026-07-05T07:50:17.370097+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.18563","created_at":"2026-07-05T07:50:17.370097+00:00"},{"alias_kind":"pith_short_12","alias_value":"5KT5NQYAVFG2","created_at":"2026-07-05T07:50:17.370097+00:00"},{"alias_kind":"pith_short_16","alias_value":"5KT5NQYAVFG2UCD3","created_at":"2026-07-05T07:50:17.370097+00:00"},{"alias_kind":"pith_short_8","alias_value":"5KT5NQYA","created_at":"2026-07-05T07:50:17.370097+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26583","citing_title":"Preference Optimization Drives Monoculture in LLM Prediction Markets","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13038","citing_title":"Nous: An Attempt to Extract and Inject the Cognition Behind Prediction-Market Behavior","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31069","citing_title":"Towards Effective Long-Video Event Prediction via Multi-Level Event Semantics Mining","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22672","citing_title":"Is Capability a Liability? More Capable Language Models Make Worse Forecasts When It Matters Most","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2405.02079","citing_title":"Argumentative Large Language Models for Explainable and Contestable Claim Verification","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22672","citing_title":"Is Capability a Liability? More Capable Language Models Make Worse Forecasts When It Matters Most","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03762","citing_title":"OracleProto: A Reproducible Framework for Benchmarking LLM Native Forecasting via Knowledge Cutoff and Temporal Masking","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03310","citing_title":"Coordination as an Architectural Layer for LLM-Based Multi-Agent Systems","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00420","citing_title":"Foresight Arena: An On-Chain Benchmark for Evaluating AI Forecasting Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18576","citing_title":"Agentic Forecasting using Sequential Bayesian Updating of Linguistic Beliefs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08606","citing_title":"Extrapolating Volition with Recursive Information Markets","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00844","citing_title":"The Oracle's Fingerprint: Correlated AI Forecasting Errors and the Limits of Bias Transmission","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15719","citing_title":"Harnessing Pre-Resolution Signals for Future Prediction Agents","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15719","citing_title":"Harnessing Pre-Resolution Signals for Future Prediction Agents","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5KT5NQYAVFG2UCD3DGDTAXOJ4C","json":"https://pith.science/pith/5KT5NQYAVFG2UCD3DGDTAXOJ4C.json","graph_json":"https://pith.science/api/pith-number/5KT5NQYAVFG2UCD3DGDTAXOJ4C/graph.json","events_json":"https://pith.science/api/pith-number/5KT5NQYAVFG2UCD3DGDTAXOJ4C/events.json","paper":"https://pith.science/paper/5KT5NQYA"},"agent_actions":{"view_html":"https://pith.science/pith/5KT5NQYAVFG2UCD3DGDTAXOJ4C","download_json":"https://pith.science/pith/5KT5NQYAVFG2UCD3DGDTAXOJ4C.json","view_paper":"https://pith.science/paper/5KT5NQYA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.18563&json=true","fetch_graph":"https://pith.science/api/pith-number/5KT5NQYAVFG2UCD3DGDTAXOJ4C/graph.json","fetch_events":"https://pith.science/api/pith-number/5KT5NQYAVFG2UCD3DGDTAXOJ4C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5KT5NQYAVFG2UCD3DGDTAXOJ4C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5KT5NQYAVFG2UCD3DGDTAXOJ4C/action/storage_attestation","attest_author":"https://pith.science/pith/5KT5NQYAVFG2UCD3DGDTAXOJ4C/action/author_attestation","sign_citation":"https://pith.science/pith/5KT5NQYAVFG2UCD3DGDTAXOJ4C/action/citation_signature","submit_replication":"https://pith.science/pith/5KT5NQYAVFG2UCD3DGDTAXOJ4C/action/replication_record"}},"created_at":"2026-07-05T07:50:17.370097+00:00","updated_at":"2026-07-05T07:50:17.370097+00:00"}