{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B2QY6IFK2LWDOTSLNRVRPI7BIS","short_pith_number":"pith:B2QY6IFK","schema_version":"1.0","canonical_sha256":"0ea18f20aad2ec374e4b6c6b17a3e144ad10c5cbfd80ecf8288ffce71970335b","source":{"kind":"arxiv","id":"2501.10674","version":2},"attestation_state":"computed","paper":{"title":"Can Multimodal LLMs do Visual Temporal Understanding and Reasoning? The answer is No!","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Alham Fikri Aji, Chenyang Lyu, Mohamed Fazli Imam","submitted_at":"2025-01-18T06:41:48Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have achieved significant advancements in tasks like Visual Question Answering (VQA) by leveraging foundational Large Language Models (LLMs). However, their abilities in specific areas such as visual temporal understanding, which is crucial for comprehending real-world dynamics, remain underexplored. To address this, we propose a challenging evaluation benchmark named TemporalVQA, consisting of two parts: 1) Temporal Order Understanding and 2) Time-lapse Estimation. The first part requires MLLMs to determine the sequence of events by analyzing temporall"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.10674","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-18T06:41:48Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"32fbd705edf6c7d5a6e848db2847ecac214b974106c212a58e76b047e98a2277","abstract_canon_sha256":"ab3c2521ae0225d9d3f42c49ff20c395d2527a253ce3f8eae070154c2cfdcdfd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:48.415743Z","signature_b64":"oKahFNEVSfSPdCaXMf5ykXBMjIsduW5F9AhN5e5faSC4wpVQF2GiqpuQTyUiv1jSPXIztk+6cHxZ5NuPWws1BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ea18f20aad2ec374e4b6c6b17a3e144ad10c5cbfd80ecf8288ffce71970335b","last_reissued_at":"2026-07-05T10:15:48.415121Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:48.415121Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Multimodal LLMs do Visual Temporal Understanding and Reasoning? The answer is No!","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Alham Fikri Aji, Chenyang Lyu, Mohamed Fazli Imam","submitted_at":"2025-01-18T06:41:48Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have achieved significant advancements in tasks like Visual Question Answering (VQA) by leveraging foundational Large Language Models (LLMs). However, their abilities in specific areas such as visual temporal understanding, which is crucial for comprehending real-world dynamics, remain underexplored. To address this, we propose a challenging evaluation benchmark named TemporalVQA, consisting of two parts: 1) Temporal Order Understanding and 2) Time-lapse Estimation. The first part requires MLLMs to determine the sequence of events by analyzing temporall"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.10674","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.10674/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.10674","created_at":"2026-07-05T10:15:48.415185+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.10674v2","created_at":"2026-07-05T10:15:48.415185+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.10674","created_at":"2026-07-05T10:15:48.415185+00:00"},{"alias_kind":"pith_short_12","alias_value":"B2QY6IFK2LWD","created_at":"2026-07-05T10:15:48.415185+00:00"},{"alias_kind":"pith_short_16","alias_value":"B2QY6IFK2LWDOTSL","created_at":"2026-07-05T10:15:48.415185+00:00"},{"alias_kind":"pith_short_8","alias_value":"B2QY6IFK","created_at":"2026-07-05T10:15:48.415185+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05702","citing_title":"Seeing Time: Benchmarking Chronological Reasoning and Shortcut Biases in Vision-Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27828","citing_title":"Video-MME-Logical: A Controlled Diagnostic Benchmark for Video Temporal-Logical Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2506.07180","citing_title":"Flattery in Motion: Benchmarking and Analyzing Sycophancy in Video-LLMs","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01333","citing_title":"OralMLLM-Bench: Evaluating Cognitive Capabilities of Multimodal Large Language Models in Dental Practice","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22829","citing_title":"Lost in the Vibrations: Vision Language Models Fail the Dynamic Gauges Test","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B2QY6IFK2LWDOTSLNRVRPI7BIS","json":"https://pith.science/pith/B2QY6IFK2LWDOTSLNRVRPI7BIS.json","graph_json":"https://pith.science/api/pith-number/B2QY6IFK2LWDOTSLNRVRPI7BIS/graph.json","events_json":"https://pith.science/api/pith-number/B2QY6IFK2LWDOTSLNRVRPI7BIS/events.json","paper":"https://pith.science/paper/B2QY6IFK"},"agent_actions":{"view_html":"https://pith.science/pith/B2QY6IFK2LWDOTSLNRVRPI7BIS","download_json":"https://pith.science/pith/B2QY6IFK2LWDOTSLNRVRPI7BIS.json","view_paper":"https://pith.science/paper/B2QY6IFK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.10674&json=true","fetch_graph":"https://pith.science/api/pith-number/B2QY6IFK2LWDOTSLNRVRPI7BIS/graph.json","fetch_events":"https://pith.science/api/pith-number/B2QY6IFK2LWDOTSLNRVRPI7BIS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B2QY6IFK2LWDOTSLNRVRPI7BIS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B2QY6IFK2LWDOTSLNRVRPI7BIS/action/storage_attestation","attest_author":"https://pith.science/pith/B2QY6IFK2LWDOTSLNRVRPI7BIS/action/author_attestation","sign_citation":"https://pith.science/pith/B2QY6IFK2LWDOTSLNRVRPI7BIS/action/citation_signature","submit_replication":"https://pith.science/pith/B2QY6IFK2LWDOTSLNRVRPI7BIS/action/replication_record"}},"created_at":"2026-07-05T10:15:48.415185+00:00","updated_at":"2026-07-05T10:15:48.415185+00:00"}