{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UBU3DY5PPHJENBA3VDIQDMWLCS","short_pith_number":"pith:UBU3DY5P","schema_version":"1.0","canonical_sha256":"a069b1e3af79d246841ba8d101b2cb14ae60a5921f80ef7d238eb0d0cf542915","source":{"kind":"arxiv","id":"2412.04814","version":3},"attestation_state":"computed","paper":{"title":"LiFT: Leveraging Human Feedback for Text-to-Video Model Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheng Jin, Hao Li, Junyan Wang, Xiaomeng Yang, Yibin Wang, Zhiyu Tan","submitted_at":"2024-12-06T07:16:14Z","abstract_excerpt":"Recent advances in text-to-video (T2V) generative models have shown impressive capabilities. However, these models are still inadequate in aligning synthesized videos with human preferences (e.g., accurately reflecting text descriptions), which is particularly difficult to address, as human preferences are subjective and challenging to formalize as objective functions. Existing studies train video quality assessment models that rely on human-annotated ratings for video evaluation but overlook the reasoning behind evaluations, limiting their ability to capture nuanced human criteria. Moreover, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.04814","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-06T07:16:14Z","cross_cats_sorted":[],"title_canon_sha256":"67411d12f10d93c18deaf5a0e9e43bf1f14fe0b689598cbb17c425676d787a30","abstract_canon_sha256":"4d35e0c7dceee620b273b086ab8f3c29971e0ecb71ae31c055dc9b29d40ee75f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:24:28.498473Z","signature_b64":"tiOxsvTOBVeQyNRQ+QKUKjgtP5F+uwawaKBd/VDMlz/ImY01lVLkBS9aPA2F3Ls2DRnwgDmGTrMXLovh14qWCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a069b1e3af79d246841ba8d101b2cb14ae60a5921f80ef7d238eb0d0cf542915","last_reissued_at":"2026-07-05T10:24:28.497478Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:24:28.497478Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LiFT: Leveraging Human Feedback for Text-to-Video Model Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheng Jin, Hao Li, Junyan Wang, Xiaomeng Yang, Yibin Wang, Zhiyu Tan","submitted_at":"2024-12-06T07:16:14Z","abstract_excerpt":"Recent advances in text-to-video (T2V) generative models have shown impressive capabilities. However, these models are still inadequate in aligning synthesized videos with human preferences (e.g., accurately reflecting text descriptions), which is particularly difficult to address, as human preferences are subjective and challenging to formalize as objective functions. Existing studies train video quality assessment models that rely on human-annotated ratings for video evaluation but overlook the reasoning behind evaluations, limiting their ability to capture nuanced human criteria. Moreover, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.04814","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.04814/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.04814","created_at":"2026-07-05T10:24:28.497602+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.04814v3","created_at":"2026-07-05T10:24:28.497602+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.04814","created_at":"2026-07-05T10:24:28.497602+00:00"},{"alias_kind":"pith_short_12","alias_value":"UBU3DY5PPHJE","created_at":"2026-07-05T10:24:28.497602+00:00"},{"alias_kind":"pith_short_16","alias_value":"UBU3DY5PPHJENBA3","created_at":"2026-07-05T10:24:28.497602+00:00"},{"alias_kind":"pith_short_8","alias_value":"UBU3DY5P","created_at":"2026-07-05T10:24:28.497602+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20310","citing_title":"Through the PRISM: Preference Representation in Intermediate States of Video Diffusion Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11723","citing_title":"CaC: Advancing Video Reward Models via Hierarchical Spatiotemporal Concentrating","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25661","citing_title":"DRM: Diffusion-based Reward Model With Step-wise Guidance","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01843","citing_title":"PhyDetEx: Detecting and Explaining the Physical Plausibility of T2V Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2601.04068","citing_title":"Mind the Generative Details: Direct Localized Detail Preference Optimization for Video Diffusion Models","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20206","citing_title":"RAPO++: Cross-Stage Prompt Optimization for Text-to-Video Generation via Data Alignment and Test-Time Scaling","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2601.04068","citing_title":"Mind the Generative Details: Direct Localized Detail Preference Optimization for Video Diffusion Models","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2503.05236","citing_title":"Unified Reward Model for Multimodal Understanding and Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13918","citing_title":"Improving Video Generation with Human Feedback","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05922","citing_title":"Think, then Score: Decoupled Reasoning and Scoring for Video Reward Modeling","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11723","citing_title":"CaC: Advancing Video Reward Models via Hierarchical Spatiotemporal Concentrating","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UBU3DY5PPHJENBA3VDIQDMWLCS","json":"https://pith.science/pith/UBU3DY5PPHJENBA3VDIQDMWLCS.json","graph_json":"https://pith.science/api/pith-number/UBU3DY5PPHJENBA3VDIQDMWLCS/graph.json","events_json":"https://pith.science/api/pith-number/UBU3DY5PPHJENBA3VDIQDMWLCS/events.json","paper":"https://pith.science/paper/UBU3DY5P"},"agent_actions":{"view_html":"https://pith.science/pith/UBU3DY5PPHJENBA3VDIQDMWLCS","download_json":"https://pith.science/pith/UBU3DY5PPHJENBA3VDIQDMWLCS.json","view_paper":"https://pith.science/paper/UBU3DY5P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.04814&json=true","fetch_graph":"https://pith.science/api/pith-number/UBU3DY5PPHJENBA3VDIQDMWLCS/graph.json","fetch_events":"https://pith.science/api/pith-number/UBU3DY5PPHJENBA3VDIQDMWLCS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UBU3DY5PPHJENBA3VDIQDMWLCS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UBU3DY5PPHJENBA3VDIQDMWLCS/action/storage_attestation","attest_author":"https://pith.science/pith/UBU3DY5PPHJENBA3VDIQDMWLCS/action/author_attestation","sign_citation":"https://pith.science/pith/UBU3DY5PPHJENBA3VDIQDMWLCS/action/citation_signature","submit_replication":"https://pith.science/pith/UBU3DY5PPHJENBA3VDIQDMWLCS/action/replication_record"}},"created_at":"2026-07-05T10:24:28.497602+00:00","updated_at":"2026-07-05T10:24:28.497602+00:00"}