{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:EWMFQBXKVTRCQDCMDFPUZXOB6H","short_pith_number":"pith:EWMFQBXK","schema_version":"1.0","canonical_sha256":"25985806eaace2280c4c195f4cddc1f1e78506c6958ce5220436756febf5e897","source":{"kind":"arxiv","id":"2310.05157","version":1},"attestation_state":"computed","paper":{"title":"MenatQA: A New Dataset for Testing the Temporal Comprehension and Reasoning Abilities of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fangyu Lei, Huanhuan Ma, Jun Zhao, Kang Liu, Xiaoyan Yu, Yifan Wei, Yisong Su, Yuanzhe Zhang","submitted_at":"2023-10-08T13:19:52Z","abstract_excerpt":"Large language models (LLMs) have shown nearly saturated performance on many natural language processing (NLP) tasks. As a result, it is natural for people to believe that LLMs have also mastered abilities such as time understanding and reasoning. However, research on the temporal sensitivity of LLMs has been insufficiently emphasized. To fill this gap, this paper constructs Multiple Sensitive Factors Time QA (MenatQA), which encompasses three temporal factors (scope factor, order factor, counterfactual factor) with total 2,853 samples for evaluating the time comprehension and reasoning abilit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.05157","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-08T13:19:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"347099fa6f153683d5bc0d644b50589779d169cd1c6a2fe702897ecae434e621","abstract_canon_sha256":"6817139268e353753b5b327e85719d6a7074b8358c9a41f4739f2d20c82dd627"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:58:27.544288Z","signature_b64":"hyqhwkPEE6K7xtFygZolJyrTBC9U7ieVzCqEVimaFOBPPdgd/7dipCdnoaE8keBR6mlRzEf2RlqTAD2jbXGyBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25985806eaace2280c4c195f4cddc1f1e78506c6958ce5220436756febf5e897","last_reissued_at":"2026-07-05T06:58:27.543821Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:58:27.543821Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MenatQA: A New Dataset for Testing the Temporal Comprehension and Reasoning Abilities of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fangyu Lei, Huanhuan Ma, Jun Zhao, Kang Liu, Xiaoyan Yu, Yifan Wei, Yisong Su, Yuanzhe Zhang","submitted_at":"2023-10-08T13:19:52Z","abstract_excerpt":"Large language models (LLMs) have shown nearly saturated performance on many natural language processing (NLP) tasks. As a result, it is natural for people to believe that LLMs have also mastered abilities such as time understanding and reasoning. However, research on the temporal sensitivity of LLMs has been insufficiently emphasized. To fill this gap, this paper constructs Multiple Sensitive Factors Time QA (MenatQA), which encompasses three temporal factors (scope factor, order factor, counterfactual factor) with total 2,853 samples for evaluating the time comprehension and reasoning abilit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.05157","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.05157/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.05157","created_at":"2026-07-05T06:58:27.543878+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.05157v1","created_at":"2026-07-05T06:58:27.543878+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.05157","created_at":"2026-07-05T06:58:27.543878+00:00"},{"alias_kind":"pith_short_12","alias_value":"EWMFQBXKVTRC","created_at":"2026-07-05T06:58:27.543878+00:00"},{"alias_kind":"pith_short_16","alias_value":"EWMFQBXKVTRCQDCM","created_at":"2026-07-05T06:58:27.543878+00:00"},{"alias_kind":"pith_short_8","alias_value":"EWMFQBXK","created_at":"2026-07-05T06:58:27.543878+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.21836","citing_title":"AutoTIR: Autonomous Tools Integrated Reasoning via Reinforcement Learning","ref_index":57,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EWMFQBXKVTRCQDCMDFPUZXOB6H","json":"https://pith.science/pith/EWMFQBXKVTRCQDCMDFPUZXOB6H.json","graph_json":"https://pith.science/api/pith-number/EWMFQBXKVTRCQDCMDFPUZXOB6H/graph.json","events_json":"https://pith.science/api/pith-number/EWMFQBXKVTRCQDCMDFPUZXOB6H/events.json","paper":"https://pith.science/paper/EWMFQBXK"},"agent_actions":{"view_html":"https://pith.science/pith/EWMFQBXKVTRCQDCMDFPUZXOB6H","download_json":"https://pith.science/pith/EWMFQBXKVTRCQDCMDFPUZXOB6H.json","view_paper":"https://pith.science/paper/EWMFQBXK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.05157&json=true","fetch_graph":"https://pith.science/api/pith-number/EWMFQBXKVTRCQDCMDFPUZXOB6H/graph.json","fetch_events":"https://pith.science/api/pith-number/EWMFQBXKVTRCQDCMDFPUZXOB6H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EWMFQBXKVTRCQDCMDFPUZXOB6H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EWMFQBXKVTRCQDCMDFPUZXOB6H/action/storage_attestation","attest_author":"https://pith.science/pith/EWMFQBXKVTRCQDCMDFPUZXOB6H/action/author_attestation","sign_citation":"https://pith.science/pith/EWMFQBXKVTRCQDCMDFPUZXOB6H/action/citation_signature","submit_replication":"https://pith.science/pith/EWMFQBXKVTRCQDCMDFPUZXOB6H/action/replication_record"}},"created_at":"2026-07-05T06:58:27.543878+00:00","updated_at":"2026-07-05T06:58:27.543878+00:00"}