{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:A3ISHNMIDGGH4KXPTWM33J4VQQ","short_pith_number":"pith:A3ISHNMI","schema_version":"1.0","canonical_sha256":"06d123b588198c7e2aef9d99bda79584307b9738f8ce2a0bb6a87c5e4cf64642","source":{"kind":"arxiv","id":"2408.11318","version":2},"attestation_state":"computed","paper":{"title":"TWLV-I: Analysis and Insights from Holistic Evaluation on Video Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aiden Lee, Daewoo Kim, GeunOh Kim, Hyeongmin Lee, Hyojun Go, Jaehyuk Yi, Jangwon Lee, Jay Suh, Jiho Jang, Jihwan Kim, Jin-Young Kim, Jongmok Kim, Jongseok Kim, Junwan Kim, Kyungjune Baek, Minjoon Seo, Raehyuk Jung, Seokjin Han, Seongsu Ha, Seungjoon Park, Soonwoo Kwon","submitted_at":"2024-08-21T03:56:27Z","abstract_excerpt":"In this work, we discuss evaluating video foundation models in a fair and robust manner. Unlike language or image foundation models, many video foundation models are evaluated with differing parameters (such as sampling rate, number of frames, pretraining steps, etc.), making fair and robust comparisons challenging. Therefore, we present a carefully designed evaluation framework for measuring two core capabilities of video comprehension: appearance and motion understanding. Our findings reveal that existing video foundation models, whether text-supervised like UMT or InternVideo2, or self-supe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.11318","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-21T03:56:27Z","cross_cats_sorted":[],"title_canon_sha256":"7ce9a24e574914264d1dc480e3af3c7f64077956e0e3d6bfa0ad7faa1789e1ed","abstract_canon_sha256":"1b3b61e912b99a5fb36462b21bfe2405616415dc6c5270f31ce6391a5c9d999b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:58:26.472494Z","signature_b64":"aJY8fIWM8vuRBMYV7sVnoPnoRfVoIj4Zyvuxu11dwvp+EX5sEzc+c0Ak5F6MAUeF7cCTTp9kcwC43Pdk5h4LCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"06d123b588198c7e2aef9d99bda79584307b9738f8ce2a0bb6a87c5e4cf64642","last_reissued_at":"2026-07-05T08:58:26.469906Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:58:26.469906Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TWLV-I: Analysis and Insights from Holistic Evaluation on Video Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aiden Lee, Daewoo Kim, GeunOh Kim, Hyeongmin Lee, Hyojun Go, Jaehyuk Yi, Jangwon Lee, Jay Suh, Jiho Jang, Jihwan Kim, Jin-Young Kim, Jongmok Kim, Jongseok Kim, Junwan Kim, Kyungjune Baek, Minjoon Seo, Raehyuk Jung, Seokjin Han, Seongsu Ha, Seungjoon Park, Soonwoo Kwon","submitted_at":"2024-08-21T03:56:27Z","abstract_excerpt":"In this work, we discuss evaluating video foundation models in a fair and robust manner. Unlike language or image foundation models, many video foundation models are evaluated with differing parameters (such as sampling rate, number of frames, pretraining steps, etc.), making fair and robust comparisons challenging. Therefore, we present a carefully designed evaluation framework for measuring two core capabilities of video comprehension: appearance and motion understanding. Our findings reveal that existing video foundation models, whether text-supervised like UMT or InternVideo2, or self-supe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.11318","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.11318/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.11318","created_at":"2026-07-05T08:58:26.470086+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.11318v2","created_at":"2026-07-05T08:58:26.470086+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.11318","created_at":"2026-07-05T08:58:26.470086+00:00"},{"alias_kind":"pith_short_12","alias_value":"A3ISHNMIDGGH","created_at":"2026-07-05T08:58:26.470086+00:00"},{"alias_kind":"pith_short_16","alias_value":"A3ISHNMIDGGH4KXP","created_at":"2026-07-05T08:58:26.470086+00:00"},{"alias_kind":"pith_short_8","alias_value":"A3ISHNMI","created_at":"2026-07-05T08:58:26.470086+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A3ISHNMIDGGH4KXPTWM33J4VQQ","json":"https://pith.science/pith/A3ISHNMIDGGH4KXPTWM33J4VQQ.json","graph_json":"https://pith.science/api/pith-number/A3ISHNMIDGGH4KXPTWM33J4VQQ/graph.json","events_json":"https://pith.science/api/pith-number/A3ISHNMIDGGH4KXPTWM33J4VQQ/events.json","paper":"https://pith.science/paper/A3ISHNMI"},"agent_actions":{"view_html":"https://pith.science/pith/A3ISHNMIDGGH4KXPTWM33J4VQQ","download_json":"https://pith.science/pith/A3ISHNMIDGGH4KXPTWM33J4VQQ.json","view_paper":"https://pith.science/paper/A3ISHNMI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.11318&json=true","fetch_graph":"https://pith.science/api/pith-number/A3ISHNMIDGGH4KXPTWM33J4VQQ/graph.json","fetch_events":"https://pith.science/api/pith-number/A3ISHNMIDGGH4KXPTWM33J4VQQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A3ISHNMIDGGH4KXPTWM33J4VQQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A3ISHNMIDGGH4KXPTWM33J4VQQ/action/storage_attestation","attest_author":"https://pith.science/pith/A3ISHNMIDGGH4KXPTWM33J4VQQ/action/author_attestation","sign_citation":"https://pith.science/pith/A3ISHNMIDGGH4KXPTWM33J4VQQ/action/citation_signature","submit_replication":"https://pith.science/pith/A3ISHNMIDGGH4KXPTWM33J4VQQ/action/replication_record"}},"created_at":"2026-07-05T08:58:26.470086+00:00","updated_at":"2026-07-05T08:58:26.470086+00:00"}