{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OY2BZDRP4JXEFLLSFXHS5KV7SS","short_pith_number":"pith:OY2BZDRP","schema_version":"1.0","canonical_sha256":"76341c8e2fe26e42ad722dcf2eaabf94a2c96f2eb28fa46b9ae80f58c444d947","source":{"kind":"arxiv","id":"2311.17982","version":1},"attestation_state":"computed","paper":{"title":"VBench: Comprehensive Benchmark Suite for Video Generative Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenyang Si, Dahua Lin, Fan Zhang, Jiashuo Yu, Limin Wang, Nattapol Chanpaisit, Qingyang Jin, Tianxing Wu, Xinyuan Chen, Yaohui Wang, Yinan He, Yuanhan Zhang, Yuming Jiang, Yu Qiao, Ziqi Huang, Ziwei Liu","submitted_at":"2023-11-29T18:39:01Z","abstract_excerpt":"Video generation has witnessed significant advancements, yet evaluating these models remains a challenge. A comprehensive evaluation benchmark for video generation is indispensable for two reasons: 1) Existing metrics do not fully align with human perceptions; 2) An ideal evaluation system should provide insights to inform future developments of video generation. To this end, we present VBench, a comprehensive benchmark suite that dissects \"video generation quality\" into specific, hierarchical, and disentangled dimensions, each with tailored prompts and evaluation methods. VBench has three app"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.17982","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-11-29T18:39:01Z","cross_cats_sorted":[],"title_canon_sha256":"9a4942f13d963f5a40a9b581add97555d5ea09b8dd01050801f2176e97fddeb5","abstract_canon_sha256":"74fdcce85ce34b9d332615c0b822d352a3b7dc00638f539d56fdddda81d62f46"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:18:34.857185Z","signature_b64":"qilcPv3wo32plgD0YSGXAOK/cZPTL116E1p+D/hmPIEtHR/uO1i57FxTSubri3tpCEdL5dQwP9knm4YgH324Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"76341c8e2fe26e42ad722dcf2eaabf94a2c96f2eb28fa46b9ae80f58c444d947","last_reissued_at":"2026-07-05T07:18:34.856651Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:18:34.856651Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VBench: Comprehensive Benchmark Suite for Video Generative Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenyang Si, Dahua Lin, Fan Zhang, Jiashuo Yu, Limin Wang, Nattapol Chanpaisit, Qingyang Jin, Tianxing Wu, Xinyuan Chen, Yaohui Wang, Yinan He, Yuanhan Zhang, Yuming Jiang, Yu Qiao, Ziqi Huang, Ziwei Liu","submitted_at":"2023-11-29T18:39:01Z","abstract_excerpt":"Video generation has witnessed significant advancements, yet evaluating these models remains a challenge. A comprehensive evaluation benchmark for video generation is indispensable for two reasons: 1) Existing metrics do not fully align with human perceptions; 2) An ideal evaluation system should provide insights to inform future developments of video generation. To this end, we present VBench, a comprehensive benchmark suite that dissects \"video generation quality\" into specific, hierarchical, and disentangled dimensions, each with tailored prompts and evaluation methods. VBench has three app"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.17982","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.17982/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.17982","created_at":"2026-07-05T07:18:34.856710+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.17982v1","created_at":"2026-07-05T07:18:34.856710+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.17982","created_at":"2026-07-05T07:18:34.856710+00:00"},{"alias_kind":"pith_short_12","alias_value":"OY2BZDRP4JXE","created_at":"2026-07-05T07:18:34.856710+00:00"},{"alias_kind":"pith_short_16","alias_value":"OY2BZDRP4JXEFLLS","created_at":"2026-07-05T07:18:34.856710+00:00"},{"alias_kind":"pith_short_8","alias_value":"OY2BZDRP","created_at":"2026-07-05T07:18:34.856710+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24225","citing_title":"Geometry-Instructed Video Editing","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02075","citing_title":"HandsOnWorld: Unconstrained Egocentric Video Generation with Camera-Disentangled Hand Control","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08091","citing_title":"VideoWeaver: Evaluating and Evolving Skills for Agentic Long Video Generation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28215","citing_title":"HAT-4D: Lifting Monocular Video for 4D Multi-Object Interactions via Human-Agent Collaboration","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26266","citing_title":"Quantized Keys Steal Attention: Bias Correction for KV-Cache Compression in Video Diffusion","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29360","citing_title":"MiraBench: Evaluating Action-Conditioned Reliability in Robotic World Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00658","citing_title":"Collaborative Few-Step Distillation and Low-Bit Quantization for Wan2.2 Dual-Expert Video Diffusion Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23699","citing_title":"CRONOS: Benchmarking Counterfactual Physical Consistency in Video Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21977","citing_title":"Video as Natural Augmentation: Towards Unified AI-Generated Image and Video Detection","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22144","citing_title":"One Sentence, One Drama: Personalized Short-Form Drama Generation via Multi-Agent Systems","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21042","citing_title":"Dynamic Video Generation: Shaping Video Generation Across Time and Space","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2406.03520","citing_title":"VideoPhy: Evaluating Physical Commonsense for Video Generation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2602.05449","citing_title":"DisCa: Accelerating Video Diffusion Transformers with Distillation-Compatible Learnable Feature Caching","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15185","citing_title":"Quantitative Video World Model Evaluation for Geometric-Consistency","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02467","citing_title":"VERTIGO: Visual Preference Optimization for Cinematic Camera Trajectory Generation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2404.02101","citing_title":"CameraCtrl: Enabling Camera Control for Text-to-Video Generation","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05898","citing_title":"Physics-Aware Video Instance Removal Benchmark","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OY2BZDRP4JXEFLLSFXHS5KV7SS","json":"https://pith.science/pith/OY2BZDRP4JXEFLLSFXHS5KV7SS.json","graph_json":"https://pith.science/api/pith-number/OY2BZDRP4JXEFLLSFXHS5KV7SS/graph.json","events_json":"https://pith.science/api/pith-number/OY2BZDRP4JXEFLLSFXHS5KV7SS/events.json","paper":"https://pith.science/paper/OY2BZDRP"},"agent_actions":{"view_html":"https://pith.science/pith/OY2BZDRP4JXEFLLSFXHS5KV7SS","download_json":"https://pith.science/pith/OY2BZDRP4JXEFLLSFXHS5KV7SS.json","view_paper":"https://pith.science/paper/OY2BZDRP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.17982&json=true","fetch_graph":"https://pith.science/api/pith-number/OY2BZDRP4JXEFLLSFXHS5KV7SS/graph.json","fetch_events":"https://pith.science/api/pith-number/OY2BZDRP4JXEFLLSFXHS5KV7SS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OY2BZDRP4JXEFLLSFXHS5KV7SS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OY2BZDRP4JXEFLLSFXHS5KV7SS/action/storage_attestation","attest_author":"https://pith.science/pith/OY2BZDRP4JXEFLLSFXHS5KV7SS/action/author_attestation","sign_citation":"https://pith.science/pith/OY2BZDRP4JXEFLLSFXHS5KV7SS/action/citation_signature","submit_replication":"https://pith.science/pith/OY2BZDRP4JXEFLLSFXHS5KV7SS/action/replication_record"}},"created_at":"2026-07-05T07:18:34.856710+00:00","updated_at":"2026-07-05T07:18:34.856710+00:00"}