{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T5NM25ZZGY57ZSSNKOMHVVGBA6","short_pith_number":"pith:T5NM25ZZ","schema_version":"1.0","canonical_sha256":"9f5acd7739363bfcca4d53987ad4c107a14d5370cc16f70c6cd9b86097bc9906","source":{"kind":"arxiv","id":"2405.18750","version":2},"attestation_state":"computed","paper":{"title":"T2V-Turbo: Breaking the Quality Bottleneck of Video Consistency Model with Mixed Reward Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiachen Li, Sugato Basu, Tsu-Jui Fu, Weixi Feng, Wenhu Chen, William Yang Wang, Xinyi Wang","submitted_at":"2024-05-29T04:26:17Z","abstract_excerpt":"Diffusion-based text-to-video (T2V) models have achieved significant success but continue to be hampered by the slow sampling speed of their iterative sampling processes. To address the challenge, consistency models have been proposed to facilitate fast inference, albeit at the cost of sample quality. In this work, we aim to break the quality bottleneck of a video consistency model (VCM) to achieve $\\textbf{both fast and high-quality video generation}$. We introduce T2V-Turbo, which integrates feedback from a mixture of differentiable reward models into the consistency distillation (CD) proces"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.18750","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-29T04:26:17Z","cross_cats_sorted":[],"title_canon_sha256":"7ffc17a0b83aef097cc73e77629d4870cd539cb2307e38e834e9cf162271c6de","abstract_canon_sha256":"4b5e35849324c3ed0b60683514de2278de9bf4f18dbb7e6e2fee4442d83df5f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:09.869824Z","signature_b64":"Sqr2F2v5VDucy16NOhB6s9Git7g5updUu2YeaSoAFfG/9nivvCooju2fiPQu6jvN3hWK2WLwcFjdO5yOGF/DBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f5acd7739363bfcca4d53987ad4c107a14d5370cc16f70c6cd9b86097bc9906","last_reissued_at":"2026-07-05T09:19:09.869393Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:09.869393Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"T2V-Turbo: Breaking the Quality Bottleneck of Video Consistency Model with Mixed Reward Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiachen Li, Sugato Basu, Tsu-Jui Fu, Weixi Feng, Wenhu Chen, William Yang Wang, Xinyi Wang","submitted_at":"2024-05-29T04:26:17Z","abstract_excerpt":"Diffusion-based text-to-video (T2V) models have achieved significant success but continue to be hampered by the slow sampling speed of their iterative sampling processes. To address the challenge, consistency models have been proposed to facilitate fast inference, albeit at the cost of sample quality. In this work, we aim to break the quality bottleneck of a video consistency model (VCM) to achieve $\\textbf{both fast and high-quality video generation}$. We introduce T2V-Turbo, which integrates feedback from a mixture of differentiable reward models into the consistency distillation (CD) proces"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.18750","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.18750/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.18750","created_at":"2026-07-05T09:19:09.869442+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.18750v2","created_at":"2026-07-05T09:19:09.869442+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.18750","created_at":"2026-07-05T09:19:09.869442+00:00"},{"alias_kind":"pith_short_12","alias_value":"T5NM25ZZGY57","created_at":"2026-07-05T09:19:09.869442+00:00"},{"alias_kind":"pith_short_16","alias_value":"T5NM25ZZGY57ZSSN","created_at":"2026-07-05T09:19:09.869442+00:00"},{"alias_kind":"pith_short_8","alias_value":"T5NM25ZZ","created_at":"2026-07-05T09:19:09.869442+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.15689","citing_title":"DOLLAR: Few-Step Video Generation via Distillation and Latent Reward Optimization","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07982","citing_title":"Geometry Forcing: Marrying Video Diffusion and 3D Representation for Consistent World Modeling","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20206","citing_title":"RAPO++: Cross-Stage Prompt Optimization for Text-to-Video Generation via Data Alignment and Test-Time Scaling","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13918","citing_title":"Improving Video Generation with Human Feedback","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18869","citing_title":"Emu3: Next-Token Prediction is All You Need","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2408.06072","citing_title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","ref_index":84,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T5NM25ZZGY57ZSSNKOMHVVGBA6","json":"https://pith.science/pith/T5NM25ZZGY57ZSSNKOMHVVGBA6.json","graph_json":"https://pith.science/api/pith-number/T5NM25ZZGY57ZSSNKOMHVVGBA6/graph.json","events_json":"https://pith.science/api/pith-number/T5NM25ZZGY57ZSSNKOMHVVGBA6/events.json","paper":"https://pith.science/paper/T5NM25ZZ"},"agent_actions":{"view_html":"https://pith.science/pith/T5NM25ZZGY57ZSSNKOMHVVGBA6","download_json":"https://pith.science/pith/T5NM25ZZGY57ZSSNKOMHVVGBA6.json","view_paper":"https://pith.science/paper/T5NM25ZZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.18750&json=true","fetch_graph":"https://pith.science/api/pith-number/T5NM25ZZGY57ZSSNKOMHVVGBA6/graph.json","fetch_events":"https://pith.science/api/pith-number/T5NM25ZZGY57ZSSNKOMHVVGBA6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T5NM25ZZGY57ZSSNKOMHVVGBA6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T5NM25ZZGY57ZSSNKOMHVVGBA6/action/storage_attestation","attest_author":"https://pith.science/pith/T5NM25ZZGY57ZSSNKOMHVVGBA6/action/author_attestation","sign_citation":"https://pith.science/pith/T5NM25ZZGY57ZSSNKOMHVVGBA6/action/citation_signature","submit_replication":"https://pith.science/pith/T5NM25ZZGY57ZSSNKOMHVVGBA6/action/replication_record"}},"created_at":"2026-07-05T09:19:09.869442+00:00","updated_at":"2026-07-05T09:19:09.869442+00:00"}