{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TNDWLB4HA3KAHIKSGIVC3ZIDVD","short_pith_number":"pith:TNDWLB4H","schema_version":"1.0","canonical_sha256":"9b4765878706d403a152322a2de503a8ebd23b9252bdd5be159ce8157ba4ba54","source":{"kind":"arxiv","id":"2310.11440","version":3},"attestation_state":"computed","paper":{"title":"EvalCrafter: Benchmarking and Evaluating Large Video Generation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haoxin Chen, Raymond Chan, Tieyong Zeng, Xiaodong Cun, Xintao Wang, Xuebo Liu, Yang Liu, Yaofang Liu, Ying Shan, Yong Zhang","submitted_at":"2023-10-17T17:50:46Z","abstract_excerpt":"The vision and language generative models have been overgrown in recent years. For video generation, various open-sourced models and public-available services have been developed to generate high-quality videos. However, these methods often use a few metrics, e.g., FVD or IS, to evaluate the performance. We argue that it is hard to judge the large conditional generative models from the simple metrics since these models are often trained on very large datasets with multi-aspect abilities. Thus, we propose a novel framework and pipeline for exhaustively evaluating the performance of the generate"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.11440","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-10-17T17:50:46Z","cross_cats_sorted":[],"title_canon_sha256":"0b335343f15ca8397c7fe81919598694d3a8892ce8832c7848c6d3ffb27cd3de","abstract_canon_sha256":"5a742cfa0a637b8a9f9522591e0cc0081cb74af02d5d9b9e0e2616a09e820a24"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:59:55.028650Z","signature_b64":"ugSm2lHKpcZKOUWyZTO1r3uV+BQZ5m/rcRkb4mLROJ6he5C4mdsNu34kpPCGfh7F4RVsWxHrbdi1fSgxUM/HBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b4765878706d403a152322a2de503a8ebd23b9252bdd5be159ce8157ba4ba54","last_reissued_at":"2026-07-05T07:59:55.028087Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:59:55.028087Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EvalCrafter: Benchmarking and Evaluating Large Video Generation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haoxin Chen, Raymond Chan, Tieyong Zeng, Xiaodong Cun, Xintao Wang, Xuebo Liu, Yang Liu, Yaofang Liu, Ying Shan, Yong Zhang","submitted_at":"2023-10-17T17:50:46Z","abstract_excerpt":"The vision and language generative models have been overgrown in recent years. For video generation, various open-sourced models and public-available services have been developed to generate high-quality videos. However, these methods often use a few metrics, e.g., FVD or IS, to evaluate the performance. We argue that it is hard to judge the large conditional generative models from the simple metrics since these models are often trained on very large datasets with multi-aspect abilities. Thus, we propose a novel framework and pipeline for exhaustively evaluating the performance of the generate"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.11440","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.11440/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.11440","created_at":"2026-07-05T07:59:55.028156+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.11440v3","created_at":"2026-07-05T07:59:55.028156+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.11440","created_at":"2026-07-05T07:59:55.028156+00:00"},{"alias_kind":"pith_short_12","alias_value":"TNDWLB4HA3KA","created_at":"2026-07-05T07:59:55.028156+00:00"},{"alias_kind":"pith_short_16","alias_value":"TNDWLB4HA3KAHIKS","created_at":"2026-07-05T07:59:55.028156+00:00"},{"alias_kind":"pith_short_8","alias_value":"TNDWLB4H","created_at":"2026-07-05T07:59:55.028156+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28531","citing_title":"A Good Talk Does not Look Like a Summary, It Teaches You! Measuring Takeaways from Paper-to-Video Talks","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20183","citing_title":"MSAVBench: Towards Comprehensive and Reliable Evaluation of Multi-Shot Audio-Video Generation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29360","citing_title":"MiraBench: Evaluating Action-Conditioned Reliability in Robotic World Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2504.17180","citing_title":"We'll Fix it in Post: Improving Text-to-Video Generation with Neuro-Symbolic Feedback","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20183","citing_title":"MSAVBench: Towards Comprehensive and Reliable Evaluation of Multi-Shot Audio-Video Generation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2407.02371","citing_title":"OpenVid-1M: A Large-Scale High-Quality Dataset for Text-to-video Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19193","citing_title":"How Far Are Video Models from True Multimodal Reasoning?","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TNDWLB4HA3KAHIKSGIVC3ZIDVD","json":"https://pith.science/pith/TNDWLB4HA3KAHIKSGIVC3ZIDVD.json","graph_json":"https://pith.science/api/pith-number/TNDWLB4HA3KAHIKSGIVC3ZIDVD/graph.json","events_json":"https://pith.science/api/pith-number/TNDWLB4HA3KAHIKSGIVC3ZIDVD/events.json","paper":"https://pith.science/paper/TNDWLB4H"},"agent_actions":{"view_html":"https://pith.science/pith/TNDWLB4HA3KAHIKSGIVC3ZIDVD","download_json":"https://pith.science/pith/TNDWLB4HA3KAHIKSGIVC3ZIDVD.json","view_paper":"https://pith.science/paper/TNDWLB4H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.11440&json=true","fetch_graph":"https://pith.science/api/pith-number/TNDWLB4HA3KAHIKSGIVC3ZIDVD/graph.json","fetch_events":"https://pith.science/api/pith-number/TNDWLB4HA3KAHIKSGIVC3ZIDVD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TNDWLB4HA3KAHIKSGIVC3ZIDVD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TNDWLB4HA3KAHIKSGIVC3ZIDVD/action/storage_attestation","attest_author":"https://pith.science/pith/TNDWLB4HA3KAHIKSGIVC3ZIDVD/action/author_attestation","sign_citation":"https://pith.science/pith/TNDWLB4HA3KAHIKSGIVC3ZIDVD/action/citation_signature","submit_replication":"https://pith.science/pith/TNDWLB4HA3KAHIKSGIVC3ZIDVD/action/replication_record"}},"created_at":"2026-07-05T07:59:55.028156+00:00","updated_at":"2026-07-05T07:59:55.028156+00:00"}