{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TDS6J7DZE6QSYCR3VD35XZSWC6","short_pith_number":"pith:TDS6J7DZ","schema_version":"1.0","canonical_sha256":"98e5e4fc7927a12c0a3ba8f7dbe656178c2429bfa49160aeffc06c82ebff5625","source":{"kind":"arxiv","id":"2406.15252","version":3},"attestation_state":"computed","paper":{"title":"VideoScore: Building Automatic Metrics to Simulate Fine-grained Human Feedback for Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Aaran Arulraj, Abhranil Chandra, Achint Soni, Bohan Lyu, Dongfu Jiang, Ge Zhang, Haonan Chen, Kai Wang, Max Ku, Quy Duc Do, Rongqi Fan, Sherman Siu, Wenhu Chen, Xuan He, Yaswanth Narsupalli, Yuansheng Ni, Yuchen Lin, Zhiheng Lyu, Ziyan Jiang","submitted_at":"2024-06-21T15:43:46Z","abstract_excerpt":"The recent years have witnessed great advances in video generation. However, the development of automatic video metrics is lagging significantly behind. None of the existing metric is able to provide reliable scores over generated videos. The main barrier is the lack of large-scale human-annotated dataset. In this paper, we release VideoFeedback, the first large-scale dataset containing human-provided multi-aspect score over 37.6K synthesized videos from 11 existing video generative models. We train VideoScore (initialized from Mantis) based on VideoFeedback to enable automatic video quality a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.15252","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-21T15:43:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1e3a3dcf74242a2492f3b510594877e8c34ed7b1771cbc67b8c068585ebaa7a3","abstract_canon_sha256":"10e7e5827ba4709ac94bcebd1bacb12c1b5cdba901536eef653084c5f115e8ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:59.991745Z","signature_b64":"2CMxdykLN/cekrM4lFSijNYqPeE0OAosVb8qGzUYyURM1jK8a3bmRTeuWo+W4fuEfFMu32nFeP9R8J/q91BfAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98e5e4fc7927a12c0a3ba8f7dbe656178c2429bfa49160aeffc06c82ebff5625","last_reissued_at":"2026-07-05T09:19:59.991245Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:59.991245Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoScore: Building Automatic Metrics to Simulate Fine-grained Human Feedback for Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Aaran Arulraj, Abhranil Chandra, Achint Soni, Bohan Lyu, Dongfu Jiang, Ge Zhang, Haonan Chen, Kai Wang, Max Ku, Quy Duc Do, Rongqi Fan, Sherman Siu, Wenhu Chen, Xuan He, Yaswanth Narsupalli, Yuansheng Ni, Yuchen Lin, Zhiheng Lyu, Ziyan Jiang","submitted_at":"2024-06-21T15:43:46Z","abstract_excerpt":"The recent years have witnessed great advances in video generation. However, the development of automatic video metrics is lagging significantly behind. None of the existing metric is able to provide reliable scores over generated videos. The main barrier is the lack of large-scale human-annotated dataset. In this paper, we release VideoFeedback, the first large-scale dataset containing human-provided multi-aspect score over 37.6K synthesized videos from 11 existing video generative models. We train VideoScore (initialized from Mantis) based on VideoFeedback to enable automatic video quality a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.15252","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.15252/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.15252","created_at":"2026-07-05T09:19:59.991304+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.15252v3","created_at":"2026-07-05T09:19:59.991304+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.15252","created_at":"2026-07-05T09:19:59.991304+00:00"},{"alias_kind":"pith_short_12","alias_value":"TDS6J7DZE6QS","created_at":"2026-07-05T09:19:59.991304+00:00"},{"alias_kind":"pith_short_16","alias_value":"TDS6J7DZE6QSYCR3","created_at":"2026-07-05T09:19:59.991304+00:00"},{"alias_kind":"pith_short_8","alias_value":"TDS6J7DZ","created_at":"2026-07-05T09:19:59.991304+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28531","citing_title":"A Good Talk Does not Look Like a Summary, It Teaches You! Measuring Takeaways from Paper-to-Video Talks","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2412.15689","citing_title":"DOLLAR: Few-Step Video Generation via Distillation and Latent Reward Optimization","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18719","citing_title":"Seeing What Matters: Visual Preference Policy Optimization for Visual Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2501.09038","citing_title":"Do generative video models understand physical principles?","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05363","citing_title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20206","citing_title":"RAPO++: Cross-Stage Prompt Optimization for Text-to-Video Generation via Data Alignment and Test-Time Scaling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18373","citing_title":"MASS: Motion-Aware Spatial-Temporal Grounding for Physics Reasoning and Comprehension in Vision-Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14269","citing_title":"PhyMotion: Structured 3D Motion Reward for Physics-Grounded Human Video Generation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2503.05236","citing_title":"Unified Reward Model for Multimodal Understanding and Generation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13918","citing_title":"Improving Video Generation with Human Feedback","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25361","citing_title":"HuM-Eval: A Coarse-to-Fine Framework for Human-Centric Video Evaluation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07818","citing_title":"DanceGRPO: Unleashing GRPO on Visual Generation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19234","citing_title":"Learning to Credit the Right Steps: Objective-aware Process Optimization for Visual Generation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19193","citing_title":"How Far Are Video Models from True Multimodal Reasoning?","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TDS6J7DZE6QSYCR3VD35XZSWC6","json":"https://pith.science/pith/TDS6J7DZE6QSYCR3VD35XZSWC6.json","graph_json":"https://pith.science/api/pith-number/TDS6J7DZE6QSYCR3VD35XZSWC6/graph.json","events_json":"https://pith.science/api/pith-number/TDS6J7DZE6QSYCR3VD35XZSWC6/events.json","paper":"https://pith.science/paper/TDS6J7DZ"},"agent_actions":{"view_html":"https://pith.science/pith/TDS6J7DZE6QSYCR3VD35XZSWC6","download_json":"https://pith.science/pith/TDS6J7DZE6QSYCR3VD35XZSWC6.json","view_paper":"https://pith.science/paper/TDS6J7DZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.15252&json=true","fetch_graph":"https://pith.science/api/pith-number/TDS6J7DZE6QSYCR3VD35XZSWC6/graph.json","fetch_events":"https://pith.science/api/pith-number/TDS6J7DZE6QSYCR3VD35XZSWC6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TDS6J7DZE6QSYCR3VD35XZSWC6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TDS6J7DZE6QSYCR3VD35XZSWC6/action/storage_attestation","attest_author":"https://pith.science/pith/TDS6J7DZE6QSYCR3VD35XZSWC6/action/author_attestation","sign_citation":"https://pith.science/pith/TDS6J7DZE6QSYCR3VD35XZSWC6/action/citation_signature","submit_replication":"https://pith.science/pith/TDS6J7DZE6QSYCR3VD35XZSWC6/action/replication_record"}},"created_at":"2026-07-05T09:19:59.991304+00:00","updated_at":"2026-07-05T09:19:59.991304+00:00"}