{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ADZFYWQOKQV62OS7RQ545RDER7","short_pith_number":"pith:ADZFYWQO","schema_version":"1.0","canonical_sha256":"00f25c5a0e542bed3a5f8c3bcec4648fcc054998e9043fbc113f1b1517fb2d83","source":{"kind":"arxiv","id":"2410.07752","version":3},"attestation_state":"computed","paper":{"title":"Lost in Time: A New Temporal Benchmark for VideoLLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cees G. M. Snoek, Daniel Cores, Manuel Mucientes, Michael Dorkenwald, Yuki M. Asano","submitted_at":"2024-10-10T09:28:36Z","abstract_excerpt":"Large language models have demonstrated impressive performance when integrated with vision models even enabling video understanding. However, evaluating video models presents its own unique challenges, for which several benchmarks have been proposed. In this paper, we show that the currently most used video-language benchmarks can be solved without requiring much temporal reasoning. We identified three main issues in existing datasets: (i) static information from single frames is often sufficient to solve the tasks (ii) the text of the questions and candidate answers is overly informative, all"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.07752","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-10T09:28:36Z","cross_cats_sorted":[],"title_canon_sha256":"78dcda207d8e635757163e932b21567c9d7689442a8d6f5b6cea6b20ea17be88","abstract_canon_sha256":"9c97c21900cea8ede69f5f4a14db05f03f675c8854e9a7bba0652da82c465004"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:36.677349Z","signature_b64":"oYXRHPWfpMbob1Ti2bbq7k2NXHTNKPlEZaJNooxjNUQLFk72PZG54JGx7XBjW4y7oY2xGBqD0XKrR3pdPP1QCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"00f25c5a0e542bed3a5f8c3bcec4648fcc054998e9043fbc113f1b1517fb2d83","last_reissued_at":"2026-07-05T10:38:36.676780Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:36.676780Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lost in Time: A New Temporal Benchmark for VideoLLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cees G. M. Snoek, Daniel Cores, Manuel Mucientes, Michael Dorkenwald, Yuki M. Asano","submitted_at":"2024-10-10T09:28:36Z","abstract_excerpt":"Large language models have demonstrated impressive performance when integrated with vision models even enabling video understanding. However, evaluating video models presents its own unique challenges, for which several benchmarks have been proposed. In this paper, we show that the currently most used video-language benchmarks can be solved without requiring much temporal reasoning. We identified three main issues in existing datasets: (i) static information from single frames is often sufficient to solve the tasks (ii) the text of the questions and candidate answers is overly informative, all"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.07752","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.07752/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.07752","created_at":"2026-07-05T10:38:36.676836+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.07752v3","created_at":"2026-07-05T10:38:36.676836+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.07752","created_at":"2026-07-05T10:38:36.676836+00:00"},{"alias_kind":"pith_short_12","alias_value":"ADZFYWQOKQV6","created_at":"2026-07-05T10:38:36.676836+00:00"},{"alias_kind":"pith_short_16","alias_value":"ADZFYWQOKQV62OS7","created_at":"2026-07-05T10:38:36.676836+00:00"},{"alias_kind":"pith_short_8","alias_value":"ADZFYWQO","created_at":"2026-07-05T10:38:36.676836+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00248","citing_title":"Seed2.0 Model Card: Towards Intelligence Frontier for Real-World Complexity","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21988","citing_title":"Learning Spatiotemporal Sensitivity in Video LLMs via Counterfactual Reinforcement Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22570","citing_title":"VGenST-Bench: A Benchmark for Spatio-Temporal Reasoning via Active Video Synthesis","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22823","citing_title":"Which Way Did It Move? Diagnosing and Overcoming Directional Motion Blindness in Video-LLMs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15342","citing_title":"Minerva-Ego: Spatiotemporal Hints for Egocentric Video Understanding","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13511","citing_title":"Adapting MLLMs for Nuanced Video Retrieval","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20633","citing_title":"Seed1.8 Model Card: Towards Generalized Real-World Agency","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23407","citing_title":"PushupBench: Your VLM is not good at counting pushups","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11399","citing_title":"Reasoning Resides in Layers: Restoring Temporal Reasoning in Video-Language Models with Layer-Selective Merging","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07062","citing_title":"Seed1.5-VL Technical Report","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07568","citing_title":"Tracing the Arrow of Time: Diagnosing Temporal Information Flow in Video-LLMs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09985","citing_title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ADZFYWQOKQV62OS7RQ545RDER7","json":"https://pith.science/pith/ADZFYWQOKQV62OS7RQ545RDER7.json","graph_json":"https://pith.science/api/pith-number/ADZFYWQOKQV62OS7RQ545RDER7/graph.json","events_json":"https://pith.science/api/pith-number/ADZFYWQOKQV62OS7RQ545RDER7/events.json","paper":"https://pith.science/paper/ADZFYWQO"},"agent_actions":{"view_html":"https://pith.science/pith/ADZFYWQOKQV62OS7RQ545RDER7","download_json":"https://pith.science/pith/ADZFYWQOKQV62OS7RQ545RDER7.json","view_paper":"https://pith.science/paper/ADZFYWQO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.07752&json=true","fetch_graph":"https://pith.science/api/pith-number/ADZFYWQOKQV62OS7RQ545RDER7/graph.json","fetch_events":"https://pith.science/api/pith-number/ADZFYWQOKQV62OS7RQ545RDER7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ADZFYWQOKQV62OS7RQ545RDER7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ADZFYWQOKQV62OS7RQ545RDER7/action/storage_attestation","attest_author":"https://pith.science/pith/ADZFYWQOKQV62OS7RQ545RDER7/action/author_attestation","sign_citation":"https://pith.science/pith/ADZFYWQOKQV62OS7RQ545RDER7/action/citation_signature","submit_replication":"https://pith.science/pith/ADZFYWQOKQV62OS7RQ545RDER7/action/replication_record"}},"created_at":"2026-07-05T10:38:36.676836+00:00","updated_at":"2026-07-05T10:38:36.676836+00:00"}