{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AOD3BX2J6627WCR6CWVZS3MWTY","short_pith_number":"pith:AOD3BX2J","schema_version":"1.0","canonical_sha256":"0387b0df49f7b5fb0a3e15ab996d969e1890295a03eb8be281afd4ff2fed17da","source":{"kind":"arxiv","id":"2411.08324","version":2},"attestation_state":"computed","paper":{"title":"Are LLMs Prescient? A Continuous Evaluation using Daily News as the Oracle","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hui Dai, Mengye Ren, Ryan Teehan","submitted_at":"2024-11-13T04:20:20Z","abstract_excerpt":"Many existing evaluation benchmarks for Large Language Models (LLMs) quickly become outdated due to the emergence of new models and training data. These benchmarks also fall short in assessing how LLM performance changes over time, as they consist of a static set of questions without a temporal dimension. To address these limitations, we propose using future event prediction as a continuous evaluation method to assess LLMs' temporal generalization and forecasting abilities. Our benchmark, Daily Oracle, automatically generates question-answer (QA) pairs from daily news, challenging LLMs to pred"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.08324","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-13T04:20:20Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e6d047b6715130d9e5dd57bb554377b460c7dac5a38e6edb666ff8e9075fa83d","abstract_canon_sha256":"1143dfba1c12f2387946a24bb7a2eb8aaa3ff3da361f9b790ee956b3efe35d16"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:33:17.382856Z","signature_b64":"xbcnTX1DrDBylyXtrfS2kttf9JVDDzwrllewajBlb7ACg3YdhMsz6seA2+5ug1EbDMTQjSyIRSwaMTM0wD6SDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0387b0df49f7b5fb0a3e15ab996d969e1890295a03eb8be281afd4ff2fed17da","last_reissued_at":"2026-07-05T11:33:17.382380Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:33:17.382380Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are LLMs Prescient? A Continuous Evaluation using Daily News as the Oracle","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hui Dai, Mengye Ren, Ryan Teehan","submitted_at":"2024-11-13T04:20:20Z","abstract_excerpt":"Many existing evaluation benchmarks for Large Language Models (LLMs) quickly become outdated due to the emergence of new models and training data. These benchmarks also fall short in assessing how LLM performance changes over time, as they consist of a static set of questions without a temporal dimension. To address these limitations, we propose using future event prediction as a continuous evaluation method to assess LLMs' temporal generalization and forecasting abilities. Our benchmark, Daily Oracle, automatically generates question-answer (QA) pairs from daily news, challenging LLMs to pred"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.08324","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.08324/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.08324","created_at":"2026-07-05T11:33:17.382441+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.08324v2","created_at":"2026-07-05T11:33:17.382441+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.08324","created_at":"2026-07-05T11:33:17.382441+00:00"},{"alias_kind":"pith_short_12","alias_value":"AOD3BX2J6627","created_at":"2026-07-05T11:33:17.382441+00:00"},{"alias_kind":"pith_short_16","alias_value":"AOD3BX2J6627WCR6","created_at":"2026-07-05T11:33:17.382441+00:00"},{"alias_kind":"pith_short_8","alias_value":"AOD3BX2J","created_at":"2026-07-05T11:33:17.382441+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21013","citing_title":"Agentic Time Machine as an Infrastructure for Future-Event Forecasting","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11816","citing_title":"WorldReasoner: Evaluating Whether Language Model Agents Forecast Events with Valid Reasoning","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AOD3BX2J6627WCR6CWVZS3MWTY","json":"https://pith.science/pith/AOD3BX2J6627WCR6CWVZS3MWTY.json","graph_json":"https://pith.science/api/pith-number/AOD3BX2J6627WCR6CWVZS3MWTY/graph.json","events_json":"https://pith.science/api/pith-number/AOD3BX2J6627WCR6CWVZS3MWTY/events.json","paper":"https://pith.science/paper/AOD3BX2J"},"agent_actions":{"view_html":"https://pith.science/pith/AOD3BX2J6627WCR6CWVZS3MWTY","download_json":"https://pith.science/pith/AOD3BX2J6627WCR6CWVZS3MWTY.json","view_paper":"https://pith.science/paper/AOD3BX2J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.08324&json=true","fetch_graph":"https://pith.science/api/pith-number/AOD3BX2J6627WCR6CWVZS3MWTY/graph.json","fetch_events":"https://pith.science/api/pith-number/AOD3BX2J6627WCR6CWVZS3MWTY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AOD3BX2J6627WCR6CWVZS3MWTY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AOD3BX2J6627WCR6CWVZS3MWTY/action/storage_attestation","attest_author":"https://pith.science/pith/AOD3BX2J6627WCR6CWVZS3MWTY/action/author_attestation","sign_citation":"https://pith.science/pith/AOD3BX2J6627WCR6CWVZS3MWTY/action/citation_signature","submit_replication":"https://pith.science/pith/AOD3BX2J6627WCR6CWVZS3MWTY/action/replication_record"}},"created_at":"2026-07-05T11:33:17.382441+00:00","updated_at":"2026-07-05T11:33:17.382441+00:00"}