{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QY3BORBOOR4UEM5LLG2TSXEYMQ","short_pith_number":"pith:QY3BORBO","schema_version":"1.0","canonical_sha256":"863617442e74794233ab59b5395c98641e32ca8716ef7dfb0a4ba66834c2e0db","source":{"kind":"arxiv","id":"2403.15377","version":4},"attestation_state":"computed","paper":{"title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baoqi Pei, Chenting Wang, Guo Chen, Hongjie Zhang, Jiashuo Yu, Jilan Xu, Kunchang Li, Limin Wang, Rongkun Zheng, Songze Li, Tianxiang Jiang, Xinhao Li, Yali Wang, Yansong Shi, Yifei Huang, Yinan He, Yi Wang, Yu Qiao, Ziang Yan, Zun Wang","submitted_at":"2024-03-22T17:57:42Z","abstract_excerpt":"We introduce InternVideo2, a new family of video foundation models (ViFM) that achieve the state-of-the-art results in video recognition, video-text tasks, and video-centric dialogue. Our core design is a progressive training approach that unifies the masked video modeling, crossmodal contrastive learning, and next token prediction, scaling up the video encoder size to 6B parameters. At the data level, we prioritize spatiotemporal consistency by semantically segmenting videos and generating video-audio-speech captions. This improves the alignment between video and text. Through extensive exper"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.15377","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-22T17:57:42Z","cross_cats_sorted":[],"title_canon_sha256":"55ffef893f23d217a6d7f4022c9ac53633fcda8b97daa97abdeaaee9df5be823","abstract_canon_sha256":"8bad424870bf9562adef298fd5f2c0ba6135fee7a3da82a5ee8949d90d3a342b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:55:22.451860Z","signature_b64":"j4AoVBFri2Wq1XR/HEhU+kfqm2zv5njxp/n05F5xBWDUcrTH/Q9lfz9KIFeaVciaAKCxJiHOYjsxVh+UmbhxCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"863617442e74794233ab59b5395c98641e32ca8716ef7dfb0a4ba66834c2e0db","last_reissued_at":"2026-07-05T08:55:22.451399Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:55:22.451399Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baoqi Pei, Chenting Wang, Guo Chen, Hongjie Zhang, Jiashuo Yu, Jilan Xu, Kunchang Li, Limin Wang, Rongkun Zheng, Songze Li, Tianxiang Jiang, Xinhao Li, Yali Wang, Yansong Shi, Yifei Huang, Yinan He, Yi Wang, Yu Qiao, Ziang Yan, Zun Wang","submitted_at":"2024-03-22T17:57:42Z","abstract_excerpt":"We introduce InternVideo2, a new family of video foundation models (ViFM) that achieve the state-of-the-art results in video recognition, video-text tasks, and video-centric dialogue. Our core design is a progressive training approach that unifies the masked video modeling, crossmodal contrastive learning, and next token prediction, scaling up the video encoder size to 6B parameters. At the data level, we prioritize spatiotemporal consistency by semantically segmenting videos and generating video-audio-speech captions. This improves the alignment between video and text. Through extensive exper"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.15377","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.15377/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.15377","created_at":"2026-07-05T08:55:22.451458+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.15377v4","created_at":"2026-07-05T08:55:22.451458+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.15377","created_at":"2026-07-05T08:55:22.451458+00:00"},{"alias_kind":"pith_short_12","alias_value":"QY3BORBOOR4U","created_at":"2026-07-05T08:55:22.451458+00:00"},{"alias_kind":"pith_short_16","alias_value":"QY3BORBOOR4UEM5L","created_at":"2026-07-05T08:55:22.451458+00:00"},{"alias_kind":"pith_short_8","alias_value":"QY3BORBO","created_at":"2026-07-05T08:55:22.451458+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06249","citing_title":"GRAMformer: Any-Order Modality Interactions via Volumetric Multimodal Cross-Attention","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03635","citing_title":"VidMsg: A Benchmark for Implicit Message Inference in Short Videos","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28215","citing_title":"HAT-4D: Lifting Monocular Video for 4D Multi-Object Interactions via Human-Agent Collaboration","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00775","citing_title":"GIRL-DETR: Gradient-Isolated Reinforcement Learning for Video Moment Retrieval","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2412.15689","citing_title":"DOLLAR: Few-Step Video Generation via Distillation and Latent Reward Optimization","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05067","citing_title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2501.02955","citing_title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16403","citing_title":"When Vision Speaks for Sound","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05363","citing_title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04590","citing_title":"VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09608","citing_title":"StreamingVLM: Real-Time Understanding for Infinite Video Streams","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17434","citing_title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12954","citing_title":"AdaFocus: Adaptive Relevance-Diversity Sampling with Zero-Cache Look-back for Efficient Long Video Understanding","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02891","citing_title":"Progressive Video Condensation with MLLM Agent for Long-form Video Understanding","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13106","citing_title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":257,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02834","citing_title":"VideoNet: A Large-Scale Dataset for Domain-Specific Action Recognition","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QY3BORBOOR4UEM5LLG2TSXEYMQ","json":"https://pith.science/pith/QY3BORBOOR4UEM5LLG2TSXEYMQ.json","graph_json":"https://pith.science/api/pith-number/QY3BORBOOR4UEM5LLG2TSXEYMQ/graph.json","events_json":"https://pith.science/api/pith-number/QY3BORBOOR4UEM5LLG2TSXEYMQ/events.json","paper":"https://pith.science/paper/QY3BORBO"},"agent_actions":{"view_html":"https://pith.science/pith/QY3BORBOOR4UEM5LLG2TSXEYMQ","download_json":"https://pith.science/pith/QY3BORBOOR4UEM5LLG2TSXEYMQ.json","view_paper":"https://pith.science/paper/QY3BORBO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.15377&json=true","fetch_graph":"https://pith.science/api/pith-number/QY3BORBOOR4UEM5LLG2TSXEYMQ/graph.json","fetch_events":"https://pith.science/api/pith-number/QY3BORBOOR4UEM5LLG2TSXEYMQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QY3BORBOOR4UEM5LLG2TSXEYMQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QY3BORBOOR4UEM5LLG2TSXEYMQ/action/storage_attestation","attest_author":"https://pith.science/pith/QY3BORBOOR4UEM5LLG2TSXEYMQ/action/author_attestation","sign_citation":"https://pith.science/pith/QY3BORBOOR4UEM5LLG2TSXEYMQ/action/citation_signature","submit_replication":"https://pith.science/pith/QY3BORBOOR4UEM5LLG2TSXEYMQ/action/replication_record"}},"created_at":"2026-07-05T08:55:22.451458+00:00","updated_at":"2026-07-05T08:55:22.451458+00:00"}