{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Z7K6SKAQZSJYYY6SNUNGASRXYD","short_pith_number":"pith:Z7K6SKAQ","schema_version":"1.0","canonical_sha256":"cfd5e92810cc938c63d26d1a604a37c0fcc9534c6252cacff5a65e72cb52665f","source":{"kind":"arxiv","id":"2406.09170","version":1},"attestation_state":"computed","paper":{"title":"Test of Time: A Benchmark for Evaluating LLMs on Temporal Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anton Tsitsulin, Bahare Fatemi, Bryan Perozzi, Jinyeong Yim, John Palowitch, Jonathan Halcrow, Karishma Malkan, Mehran Kazemi, Sungyong Seo","submitted_at":"2024-06-13T14:31:19Z","abstract_excerpt":"Large language models (LLMs) have showcased remarkable reasoning capabilities, yet they remain susceptible to errors, particularly in temporal reasoning tasks involving complex temporal logic. Existing research has explored LLM performance on temporal reasoning using diverse datasets and benchmarks. However, these studies often rely on real-world data that LLMs may have encountered during pre-training or employ anonymization techniques that can inadvertently introduce factual inconsistencies. In this work, we address these limitations by introducing novel synthetic datasets specifically design"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.09170","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-13T14:31:19Z","cross_cats_sorted":[],"title_canon_sha256":"a70e9e050823467428c06dd624f779011b03056cbf64d7bb3482df3c1d0e7ac5","abstract_canon_sha256":"63cd8f7c22b589fb13356b5ebf6eddc8c2c385669538424fb8abd9430a9e8641"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:29.478368Z","signature_b64":"p1Jn0ggGYyq6YvOtGX8ttZGM1Jt11wSwUym9u/CqgwaLFUk6TKB8RcxDq2TPtM0KsppGBMk6Cob0TVjGCo6uBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cfd5e92810cc938c63d26d1a604a37c0fcc9534c6252cacff5a65e72cb52665f","last_reissued_at":"2026-07-05T08:31:29.477846Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:29.477846Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Test of Time: A Benchmark for Evaluating LLMs on Temporal Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anton Tsitsulin, Bahare Fatemi, Bryan Perozzi, Jinyeong Yim, John Palowitch, Jonathan Halcrow, Karishma Malkan, Mehran Kazemi, Sungyong Seo","submitted_at":"2024-06-13T14:31:19Z","abstract_excerpt":"Large language models (LLMs) have showcased remarkable reasoning capabilities, yet they remain susceptible to errors, particularly in temporal reasoning tasks involving complex temporal logic. Existing research has explored LLM performance on temporal reasoning using diverse datasets and benchmarks. However, these studies often rely on real-world data that LLMs may have encountered during pre-training or employ anonymization techniques that can inadvertently introduce factual inconsistencies. In this work, we address these limitations by introducing novel synthetic datasets specifically design"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.09170","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.09170/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.09170","created_at":"2026-07-05T08:31:29.477904+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.09170v1","created_at":"2026-07-05T08:31:29.477904+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.09170","created_at":"2026-07-05T08:31:29.477904+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z7K6SKAQZSJY","created_at":"2026-07-05T08:31:29.477904+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z7K6SKAQZSJYYY6S","created_at":"2026-07-05T08:31:29.477904+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z7K6SKAQ","created_at":"2026-07-05T08:31:29.477904+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20959","citing_title":"Right Knowledge, Wrong Answer: Characterizing Parametric Temporal Conflict in Open-Weight Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05194","citing_title":"Temporal Preference Concepts and their Functions in a Large Language Model","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27378","citing_title":"Formalizing Latent Thoughts: Four Axioms of Thought Representation in LLMs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25180","citing_title":"DateSAT: A Framework for Solving Date and Period Constraints","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2503.19786","citing_title":"Gemma 3 Technical Report","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18380","citing_title":"QSTRBench: a New Benchmark to Evaluate the Ability of Language Models to Reason with Qualitative Spatial and Temporal Calculi","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08766","citing_title":"UserGPT Technical Report","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z7K6SKAQZSJYYY6SNUNGASRXYD","json":"https://pith.science/pith/Z7K6SKAQZSJYYY6SNUNGASRXYD.json","graph_json":"https://pith.science/api/pith-number/Z7K6SKAQZSJYYY6SNUNGASRXYD/graph.json","events_json":"https://pith.science/api/pith-number/Z7K6SKAQZSJYYY6SNUNGASRXYD/events.json","paper":"https://pith.science/paper/Z7K6SKAQ"},"agent_actions":{"view_html":"https://pith.science/pith/Z7K6SKAQZSJYYY6SNUNGASRXYD","download_json":"https://pith.science/pith/Z7K6SKAQZSJYYY6SNUNGASRXYD.json","view_paper":"https://pith.science/paper/Z7K6SKAQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.09170&json=true","fetch_graph":"https://pith.science/api/pith-number/Z7K6SKAQZSJYYY6SNUNGASRXYD/graph.json","fetch_events":"https://pith.science/api/pith-number/Z7K6SKAQZSJYYY6SNUNGASRXYD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z7K6SKAQZSJYYY6SNUNGASRXYD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z7K6SKAQZSJYYY6SNUNGASRXYD/action/storage_attestation","attest_author":"https://pith.science/pith/Z7K6SKAQZSJYYY6SNUNGASRXYD/action/author_attestation","sign_citation":"https://pith.science/pith/Z7K6SKAQZSJYYY6SNUNGASRXYD/action/citation_signature","submit_replication":"https://pith.science/pith/Z7K6SKAQZSJYYY6SNUNGASRXYD/action/replication_record"}},"created_at":"2026-07-05T08:31:29.477904+00:00","updated_at":"2026-07-05T08:31:29.477904+00:00"}