{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:W3L65D5SL5R46JSPFIMWQSW4RJ","short_pith_number":"pith:W3L65D5S","schema_version":"1.0","canonical_sha256":"b6d7ee8fb25f63cf264f2a19684adc8a4d9e86fbb32d91fd6f8f5e9fce753e20","source":{"kind":"arxiv","id":"2410.23266","version":2},"attestation_state":"computed","paper":{"title":"TOMATO: Assessing Visual Temporal Reasoning Capabilities in Multimodal Foundation Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Arman Cohan, Chuhan Li, Tesca FItzgerald, Yanan Zheng, Yilun Zhao, Yuxuan Ding, Ziyao Shangguan","submitted_at":"2024-10-30T17:50:23Z","abstract_excerpt":"Existing benchmarks often highlight the remarkable performance achieved by state-of-the-art Multimodal Foundation Models (MFMs) in leveraging temporal context for video understanding. However, how well do the models truly perform visual temporal reasoning? Our study of existing benchmarks shows that this capability of MFMs is likely overestimated as many questions can be solved by using a single, few, or out-of-order frames. To systematically examine current visual temporal reasoning tasks, we propose three principles with corresponding metrics: (1) Multi-Frame Gain, (2) Frame Order Sensitivit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.23266","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-30T17:50:23Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"e0b5cc64c00f696e1aba64d987c9e9a978cbb66a30f417952a39439591a34354","abstract_canon_sha256":"8a88eedcb0c30c158d426ac7f5e58843c106962512f2b5f449032fee808a5a45"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:58:21.516634Z","signature_b64":"9CZUmUBRADkdZq0FsrnqasCa4MwEArQo1oEkKcck1y33lyN4ekNY3FLTBfn+tm+gvSLmVIUNsRArYrCSoG+HBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6d7ee8fb25f63cf264f2a19684adc8a4d9e86fbb32d91fd6f8f5e9fce753e20","last_reissued_at":"2026-07-05T11:58:21.516136Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:58:21.516136Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TOMATO: Assessing Visual Temporal Reasoning Capabilities in Multimodal Foundation Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Arman Cohan, Chuhan Li, Tesca FItzgerald, Yanan Zheng, Yilun Zhao, Yuxuan Ding, Ziyao Shangguan","submitted_at":"2024-10-30T17:50:23Z","abstract_excerpt":"Existing benchmarks often highlight the remarkable performance achieved by state-of-the-art Multimodal Foundation Models (MFMs) in leveraging temporal context for video understanding. However, how well do the models truly perform visual temporal reasoning? Our study of existing benchmarks shows that this capability of MFMs is likely overestimated as many questions can be solved by using a single, few, or out-of-order frames. To systematically examine current visual temporal reasoning tasks, we propose three principles with corresponding metrics: (1) Multi-Frame Gain, (2) Frame Order Sensitivit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.23266","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.23266/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.23266","created_at":"2026-07-05T11:58:21.516203+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.23266v2","created_at":"2026-07-05T11:58:21.516203+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.23266","created_at":"2026-07-05T11:58:21.516203+00:00"},{"alias_kind":"pith_short_12","alias_value":"W3L65D5SL5R4","created_at":"2026-07-05T11:58:21.516203+00:00"},{"alias_kind":"pith_short_16","alias_value":"W3L65D5SL5R46JSP","created_at":"2026-07-05T11:58:21.516203+00:00"},{"alias_kind":"pith_short_8","alias_value":"W3L65D5S","created_at":"2026-07-05T11:58:21.516203+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":205,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13803","citing_title":"EvoGround: Self-Evolving Video Agents for Video Temporal Grounding","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26565","citing_title":"DenseStep2M: A Scalable, Training-Free Pipeline for Dense Instructional Video Annotation","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03351","citing_title":"VLMaxxing through FrameMogging Training-Free Anti-Recomputation for Video Vision-Language Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11399","citing_title":"Reasoning Resides in Layers: Restoring Temporal Reasoning in Video-Language Models with Layer-Selective Merging","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07062","citing_title":"Seed1.5-VL Technical Report","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09985","citing_title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W3L65D5SL5R46JSPFIMWQSW4RJ","json":"https://pith.science/pith/W3L65D5SL5R46JSPFIMWQSW4RJ.json","graph_json":"https://pith.science/api/pith-number/W3L65D5SL5R46JSPFIMWQSW4RJ/graph.json","events_json":"https://pith.science/api/pith-number/W3L65D5SL5R46JSPFIMWQSW4RJ/events.json","paper":"https://pith.science/paper/W3L65D5S"},"agent_actions":{"view_html":"https://pith.science/pith/W3L65D5SL5R46JSPFIMWQSW4RJ","download_json":"https://pith.science/pith/W3L65D5SL5R46JSPFIMWQSW4RJ.json","view_paper":"https://pith.science/paper/W3L65D5S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.23266&json=true","fetch_graph":"https://pith.science/api/pith-number/W3L65D5SL5R46JSPFIMWQSW4RJ/graph.json","fetch_events":"https://pith.science/api/pith-number/W3L65D5SL5R46JSPFIMWQSW4RJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W3L65D5SL5R46JSPFIMWQSW4RJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W3L65D5SL5R46JSPFIMWQSW4RJ/action/storage_attestation","attest_author":"https://pith.science/pith/W3L65D5SL5R46JSPFIMWQSW4RJ/action/author_attestation","sign_citation":"https://pith.science/pith/W3L65D5SL5R46JSPFIMWQSW4RJ/action/citation_signature","submit_replication":"https://pith.science/pith/W3L65D5SL5R46JSPFIMWQSW4RJ/action/replication_record"}},"created_at":"2026-07-05T11:58:21.516203+00:00","updated_at":"2026-07-05T11:58:21.516203+00:00"}