{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LB2FRRFU4U5BXLN7DO53LIENGT","short_pith_number":"pith:LB2FRRFU","schema_version":"1.0","canonical_sha256":"587458c4b4e53a1badbf1bbbb5a08d34dc358e76b3805ad56b84819c8b9d4a80","source":{"kind":"arxiv","id":"2409.19839","version":5},"attestation_state":"computed","paper":{"title":"ForecastBench: A Dynamic Benchmark of AI Forecasting Capabilities","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chen Yueh-Han, Danny Halawi, Ezra Karger, Fred Zhang, Houtan Bastani, Philip E. Tetlock, Zachary Jacobs","submitted_at":"2024-09-30T00:41:51Z","abstract_excerpt":"Forecasts of future events are essential inputs into informed decision-making. Machine learning (ML) systems have the potential to deliver forecasts at scale, but there is no framework for evaluating the accuracy of ML systems on a standardized set of forecasting questions. To address this gap, we introduce ForecastBench: a dynamic benchmark that evaluates the accuracy of ML systems on an automatically generated and regularly updated set of 1,000 forecasting questions. To avoid any possibility of data leakage, ForecastBench is comprised solely of questions about future events that have no know"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.19839","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-30T00:41:51Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"1ee80f602a2e8770bf7093684f4ba4efc76ea6cb55823697598195fbe3eba58e","abstract_canon_sha256":"9077631a6bdeaf67a2a9bddae3290b1c4c9a651b1034817e66cfbec0635cbd9f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:23.388552Z","signature_b64":"cZAxK2qbHuILKpI/RfkmQScqxRIAspbdO5IBhaBv9BlagRm63Obe2KrBsdHz+bquLb99ZpmC6MWrCZj7UtlIBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"587458c4b4e53a1badbf1bbbb5a08d34dc358e76b3805ad56b84819c8b9d4a80","last_reissued_at":"2026-07-05T10:21:23.388059Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:23.388059Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ForecastBench: A Dynamic Benchmark of AI Forecasting Capabilities","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chen Yueh-Han, Danny Halawi, Ezra Karger, Fred Zhang, Houtan Bastani, Philip E. Tetlock, Zachary Jacobs","submitted_at":"2024-09-30T00:41:51Z","abstract_excerpt":"Forecasts of future events are essential inputs into informed decision-making. Machine learning (ML) systems have the potential to deliver forecasts at scale, but there is no framework for evaluating the accuracy of ML systems on a standardized set of forecasting questions. To address this gap, we introduce ForecastBench: a dynamic benchmark that evaluates the accuracy of ML systems on an automatically generated and regularly updated set of 1,000 forecasting questions. To avoid any possibility of data leakage, ForecastBench is comprised solely of questions about future events that have no know"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.19839","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.19839/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.19839","created_at":"2026-07-05T10:21:23.388115+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.19839v5","created_at":"2026-07-05T10:21:23.388115+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.19839","created_at":"2026-07-05T10:21:23.388115+00:00"},{"alias_kind":"pith_short_12","alias_value":"LB2FRRFU4U5B","created_at":"2026-07-05T10:21:23.388115+00:00"},{"alias_kind":"pith_short_16","alias_value":"LB2FRRFU4U5BXLN7","created_at":"2026-07-05T10:21:23.388115+00:00"},{"alias_kind":"pith_short_8","alias_value":"LB2FRRFU","created_at":"2026-07-05T10:21:23.388115+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24996","citing_title":"From Forecasting Leaderboards to Deployment Decisions: A Fail-Closed Certification Protocol","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18686","citing_title":"ForecastBench-Sim: A Simulated-World Forecasting Benchmark","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01661","citing_title":"Diverse Evidence, Better Forecasts: Multi-Agent Deliberation Under Information Asymmetry","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00164","citing_title":"Verifiable Rewards for Calibrated Probabilistic Forecasting","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26074","citing_title":"StakeBench: Evaluating Language Understanding Grounded in Market Commitment","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14199","citing_title":"PolyBench: Benchmarking LLM Forecasting and Trading Capabilities on Live Prediction Market Data","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03762","citing_title":"OracleProto: A Reproducible Framework for Benchmarking LLM Native Forecasting via Knowledge Cutoff and Temporal Masking","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03310","citing_title":"Coordination as an Architectural Layer for LLM-Based Multi-Agent Systems","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00420","citing_title":"Foresight Arena: An On-Chain Benchmark for Evaluating AI Forecasting Agents","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16742","citing_title":"CT Open: An Open-Access, Uncontaminated, Live Platform for the Open Challenge of Clinical Trial Outcome Prediction","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24705","citing_title":"Energy-Arena: A Dynamic Benchmark for Operational Energy Forecasting","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LB2FRRFU4U5BXLN7DO53LIENGT","json":"https://pith.science/pith/LB2FRRFU4U5BXLN7DO53LIENGT.json","graph_json":"https://pith.science/api/pith-number/LB2FRRFU4U5BXLN7DO53LIENGT/graph.json","events_json":"https://pith.science/api/pith-number/LB2FRRFU4U5BXLN7DO53LIENGT/events.json","paper":"https://pith.science/paper/LB2FRRFU"},"agent_actions":{"view_html":"https://pith.science/pith/LB2FRRFU4U5BXLN7DO53LIENGT","download_json":"https://pith.science/pith/LB2FRRFU4U5BXLN7DO53LIENGT.json","view_paper":"https://pith.science/paper/LB2FRRFU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.19839&json=true","fetch_graph":"https://pith.science/api/pith-number/LB2FRRFU4U5BXLN7DO53LIENGT/graph.json","fetch_events":"https://pith.science/api/pith-number/LB2FRRFU4U5BXLN7DO53LIENGT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LB2FRRFU4U5BXLN7DO53LIENGT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LB2FRRFU4U5BXLN7DO53LIENGT/action/storage_attestation","attest_author":"https://pith.science/pith/LB2FRRFU4U5BXLN7DO53LIENGT/action/author_attestation","sign_citation":"https://pith.science/pith/LB2FRRFU4U5BXLN7DO53LIENGT/action/citation_signature","submit_replication":"https://pith.science/pith/LB2FRRFU4U5BXLN7DO53LIENGT/action/replication_record"}},"created_at":"2026-07-05T10:21:23.388115+00:00","updated_at":"2026-07-05T10:21:23.388115+00:00"}