{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6SBCXXELREEV6YR3HWEFE3WV56","short_pith_number":"pith:6SBCXXEL","schema_version":"1.0","canonical_sha256":"f4822bdc8b89095f623b3d88526ed5ef9f079dbc3621cca09a7af5163b96c733","source":{"kind":"arxiv","id":"2506.09229","version":2},"attestation_state":"computed","paper":{"title":"Cross-Frame Representation Alignment for Fine-Tuning Video Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hyojin Jang, Jaegul Choo, Kinam Kim, Minho Park, Sungwon Hwang","submitted_at":"2025-06-10T20:34:47Z","abstract_excerpt":"Fine-tuning Video Diffusion Models (VDMs) at the user level to generate videos that reflect specific attributes of training data presents notable challenges, yet remains underexplored despite its practical importance. Meanwhile, recent work such as Representation Alignment (REPA) has shown promise in improving the convergence and quality of DiT-based image diffusion models by aligning, or assimilating, its internal hidden states with external pretrained visual features, suggesting its potential for VDM fine-tuning. In this work, we first propose a straightforward adaptation of REPA for VDMs an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09229","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-10T20:34:47Z","cross_cats_sorted":[],"title_canon_sha256":"c707e32a5c5c2080372fb34abe93d36869b2012faa3d8281e784cc0e96ec4a7b","abstract_canon_sha256":"6b8c58507c1d8bcb12036fab43c47c54a064b71190ad7dece686c741bb019451"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:26:52.594031Z","signature_b64":"ypeDWZgfDP0+Vk8NFoxZ6Z7r1sOR9xP6zxGvHUEUH16mn79zWjuZ963WaZ9hGfzU0dhP7ogf7MIl17QGt44oDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f4822bdc8b89095f623b3d88526ed5ef9f079dbc3621cca09a7af5163b96c733","last_reissued_at":"2026-07-05T11:26:52.593558Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:26:52.593558Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cross-Frame Representation Alignment for Fine-Tuning Video Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hyojin Jang, Jaegul Choo, Kinam Kim, Minho Park, Sungwon Hwang","submitted_at":"2025-06-10T20:34:47Z","abstract_excerpt":"Fine-tuning Video Diffusion Models (VDMs) at the user level to generate videos that reflect specific attributes of training data presents notable challenges, yet remains underexplored despite its practical importance. Meanwhile, recent work such as Representation Alignment (REPA) has shown promise in improving the convergence and quality of DiT-based image diffusion models by aligning, or assimilating, its internal hidden states with external pretrained visual features, suggesting its potential for VDM fine-tuning. In this work, we first propose a straightforward adaptation of REPA for VDMs an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09229","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09229/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09229","created_at":"2026-07-05T11:26:52.593621+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09229v2","created_at":"2026-07-05T11:26:52.593621+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09229","created_at":"2026-07-05T11:26:52.593621+00:00"},{"alias_kind":"pith_short_12","alias_value":"6SBCXXELREEV","created_at":"2026-07-05T11:26:52.593621+00:00"},{"alias_kind":"pith_short_16","alias_value":"6SBCXXELREEV6YR3","created_at":"2026-07-05T11:26:52.593621+00:00"},{"alias_kind":"pith_short_8","alias_value":"6SBCXXEL","created_at":"2026-07-05T11:26:52.593621+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12217","citing_title":"Making Foresight Actionable: Repurposing Representation Alignment in World Action Models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01896","citing_title":"Divide and Conquer: Decoupled Representation Alignment for Multimodal World Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24962","citing_title":"Tempered Self-Similarity Alignment for Physically Plausible Video Generation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01896","citing_title":"Divide and Conquer: Decoupled Representation Alignment for Multimodal World Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11707","citing_title":"Representations Before Pixels: Semantics-Guided Hierarchical Video Prediction","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6SBCXXELREEV6YR3HWEFE3WV56","json":"https://pith.science/pith/6SBCXXELREEV6YR3HWEFE3WV56.json","graph_json":"https://pith.science/api/pith-number/6SBCXXELREEV6YR3HWEFE3WV56/graph.json","events_json":"https://pith.science/api/pith-number/6SBCXXELREEV6YR3HWEFE3WV56/events.json","paper":"https://pith.science/paper/6SBCXXEL"},"agent_actions":{"view_html":"https://pith.science/pith/6SBCXXELREEV6YR3HWEFE3WV56","download_json":"https://pith.science/pith/6SBCXXELREEV6YR3HWEFE3WV56.json","view_paper":"https://pith.science/paper/6SBCXXEL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09229&json=true","fetch_graph":"https://pith.science/api/pith-number/6SBCXXELREEV6YR3HWEFE3WV56/graph.json","fetch_events":"https://pith.science/api/pith-number/6SBCXXELREEV6YR3HWEFE3WV56/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6SBCXXELREEV6YR3HWEFE3WV56/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6SBCXXELREEV6YR3HWEFE3WV56/action/storage_attestation","attest_author":"https://pith.science/pith/6SBCXXELREEV6YR3HWEFE3WV56/action/author_attestation","sign_citation":"https://pith.science/pith/6SBCXXELREEV6YR3HWEFE3WV56/action/citation_signature","submit_replication":"https://pith.science/pith/6SBCXXELREEV6YR3HWEFE3WV56/action/replication_record"}},"created_at":"2026-07-05T11:26:52.593621+00:00","updated_at":"2026-07-05T11:26:52.593621+00:00"}