{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3JGPSU6GLXZ7UKKXWDFD4FY3I3","short_pith_number":"pith:3JGPSU6G","schema_version":"1.0","canonical_sha256":"da4cf953c65df3fa2957b0ca3e171b46d14ba78c43cc29e910a82105a68147ec","source":{"kind":"arxiv","id":"2502.05092","version":2},"attestation_state":"computed","paper":{"title":"Lost in Time: Clock and Calendar Understanding Challenges in Multimodal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Aryo Pradipta Gema, Pasquale Minervini, Rohit Saxena","submitted_at":"2025-02-07T17:11:23Z","abstract_excerpt":"Understanding time from visual representations is a fundamental cognitive skill, yet it remains a challenge for multimodal large language models (MLLMs). In this work, we investigate the capabilities of MLLMs in interpreting time and date through analogue clocks and yearly calendars. To facilitate this, we curated a structured dataset comprising two subsets: 1) $\\textit{ClockQA}$, which comprises various types of clock styles$-$standard, black-dial, no-second-hand, Roman numeral, and arrow-hand clocks$-$paired with time related questions; and 2) $\\textit{CalendarQA}$, which consists of yearly "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.05092","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-07T17:11:23Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"f53d5483075ec61db39943e2922538dfcb9740068220dbd11437021d8afef4d3","abstract_canon_sha256":"da3c6ac5b9f1f895114166f2624c8a389df99a33ef5e2b8cfa92d7a1196b4aa7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:33:44.119091Z","signature_b64":"pVPcoSx1tEQSuty+E87K89GxIULY34sbxz6Op6YLaszYWTrzuVeoAfNT6WBr6FRxGnMJ1G1FKnh/cuDdWGDfAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da4cf953c65df3fa2957b0ca3e171b46d14ba78c43cc29e910a82105a68147ec","last_reissued_at":"2026-07-05T10:33:44.118146Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:33:44.118146Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lost in Time: Clock and Calendar Understanding Challenges in Multimodal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Aryo Pradipta Gema, Pasquale Minervini, Rohit Saxena","submitted_at":"2025-02-07T17:11:23Z","abstract_excerpt":"Understanding time from visual representations is a fundamental cognitive skill, yet it remains a challenge for multimodal large language models (MLLMs). In this work, we investigate the capabilities of MLLMs in interpreting time and date through analogue clocks and yearly calendars. To facilitate this, we curated a structured dataset comprising two subsets: 1) $\\textit{ClockQA}$, which comprises various types of clock styles$-$standard, black-dial, no-second-hand, Roman numeral, and arrow-hand clocks$-$paired with time related questions; and 2) $\\textit{CalendarQA}$, which consists of yearly "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.05092","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.05092/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.05092","created_at":"2026-07-05T10:33:44.118272+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.05092v2","created_at":"2026-07-05T10:33:44.118272+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.05092","created_at":"2026-07-05T10:33:44.118272+00:00"},{"alias_kind":"pith_short_12","alias_value":"3JGPSU6GLXZ7","created_at":"2026-07-05T10:33:44.118272+00:00"},{"alias_kind":"pith_short_16","alias_value":"3JGPSU6GLXZ7UKKX","created_at":"2026-07-05T10:33:44.118272+00:00"},{"alias_kind":"pith_short_8","alias_value":"3JGPSU6G","created_at":"2026-07-05T10:33:44.118272+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08317","citing_title":"Blind-Spots-Bench: Evaluating Blind Spots in Multimodal Models","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2605.09883","citing_title":"The Cartesian Shortcut: Re-evaluate Vision Reasoning in Polar Coordinate Space","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26614","citing_title":"State Beyond Appearance: Diagnosing and Improving State Consistency in Dial-Based Measurement Reading","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09883","citing_title":"The Cartesian Shortcut: Re-evaluate Vision Reasoning in Polar Coordinate Space","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3JGPSU6GLXZ7UKKXWDFD4FY3I3","json":"https://pith.science/pith/3JGPSU6GLXZ7UKKXWDFD4FY3I3.json","graph_json":"https://pith.science/api/pith-number/3JGPSU6GLXZ7UKKXWDFD4FY3I3/graph.json","events_json":"https://pith.science/api/pith-number/3JGPSU6GLXZ7UKKXWDFD4FY3I3/events.json","paper":"https://pith.science/paper/3JGPSU6G"},"agent_actions":{"view_html":"https://pith.science/pith/3JGPSU6GLXZ7UKKXWDFD4FY3I3","download_json":"https://pith.science/pith/3JGPSU6GLXZ7UKKXWDFD4FY3I3.json","view_paper":"https://pith.science/paper/3JGPSU6G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.05092&json=true","fetch_graph":"https://pith.science/api/pith-number/3JGPSU6GLXZ7UKKXWDFD4FY3I3/graph.json","fetch_events":"https://pith.science/api/pith-number/3JGPSU6GLXZ7UKKXWDFD4FY3I3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3JGPSU6GLXZ7UKKXWDFD4FY3I3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3JGPSU6GLXZ7UKKXWDFD4FY3I3/action/storage_attestation","attest_author":"https://pith.science/pith/3JGPSU6GLXZ7UKKXWDFD4FY3I3/action/author_attestation","sign_citation":"https://pith.science/pith/3JGPSU6GLXZ7UKKXWDFD4FY3I3/action/citation_signature","submit_replication":"https://pith.science/pith/3JGPSU6GLXZ7UKKXWDFD4FY3I3/action/replication_record"}},"created_at":"2026-07-05T10:33:44.118272+00:00","updated_at":"2026-07-05T10:33:44.118272+00:00"}