{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LNS3Y4ZIRYLB3YVR3JNHYKA2K4","short_pith_number":"pith:LNS3Y4ZI","schema_version":"1.0","canonical_sha256":"5b65bc73288e161de2b1da5a7c281a5729689b5b24fbfbc8f54056b5efd9119f","source":{"kind":"arxiv","id":"2502.02492","version":2},"attestation_state":"computed","paper":{"title":"VideoJAM: Joint Appearance-Motion Representations for Enhanced Motion Generation in Video Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Adam Polyak, Amit Zohar, Hila Chefer, Lior Wolf, Shelly Sheynin, Uriel Singer, Yaniv Taigman, Yuval Kirstain","submitted_at":"2025-02-04T17:07:10Z","abstract_excerpt":"Despite tremendous recent progress, generative video models still struggle to capture real-world motion, dynamics, and physics. We show that this limitation arises from the conventional pixel reconstruction objective, which biases models toward appearance fidelity at the expense of motion coherence. To address this, we introduce VideoJAM, a novel framework that instills an effective motion prior to video generators, by encouraging the model to learn a joint appearance-motion representation. VideoJAM is composed of two complementary units. During training, we extend the objective to predict bot"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.02492","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-04T17:07:10Z","cross_cats_sorted":[],"title_canon_sha256":"2437729621c5a85e73ac6fd845b1a7be72be49993db1414d7e3816160378e304","abstract_canon_sha256":"a359befc54ce102a28e88acd8b13a21e384f4a5da7769e73b04028e2d8bebda3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:44.969133Z","signature_b64":"KmsjJoacauIcQnSpDT5SvWgk8y5FkeP357Xz7mHEU28ek0w3o45R5Bs6S/OQeLnpGihnxN0OxBD2nuecEdoPDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b65bc73288e161de2b1da5a7c281a5729689b5b24fbfbc8f54056b5efd9119f","last_reissued_at":"2026-07-05T11:09:44.968634Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:44.968634Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoJAM: Joint Appearance-Motion Representations for Enhanced Motion Generation in Video Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Adam Polyak, Amit Zohar, Hila Chefer, Lior Wolf, Shelly Sheynin, Uriel Singer, Yaniv Taigman, Yuval Kirstain","submitted_at":"2025-02-04T17:07:10Z","abstract_excerpt":"Despite tremendous recent progress, generative video models still struggle to capture real-world motion, dynamics, and physics. We show that this limitation arises from the conventional pixel reconstruction objective, which biases models toward appearance fidelity at the expense of motion coherence. To address this, we introduce VideoJAM, a novel framework that instills an effective motion prior to video generators, by encouraging the model to learn a joint appearance-motion representation. VideoJAM is composed of two complementary units. During training, we extend the objective to predict bot"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02492","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.02492/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.02492","created_at":"2026-07-05T11:09:44.968697+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.02492v2","created_at":"2026-07-05T11:09:44.968697+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02492","created_at":"2026-07-05T11:09:44.968697+00:00"},{"alias_kind":"pith_short_12","alias_value":"LNS3Y4ZIRYLB","created_at":"2026-07-05T11:09:44.968697+00:00"},{"alias_kind":"pith_short_16","alias_value":"LNS3Y4ZIRYLB3YVR","created_at":"2026-07-05T11:09:44.968697+00:00"},{"alias_kind":"pith_short_8","alias_value":"LNS3Y4ZI","created_at":"2026-07-05T11:09:44.968697+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15015","citing_title":"NEXUS: Neural Energy Fields for Physically Consistent Contact-Rich 3D Object Dynamics","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11969","citing_title":"SpecLoR: Spectral Lookahead Rectification for Motion-Coherent Text-to-Video Generation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02441","citing_title":"Spatial-Temporal Decoupled Reference Conditioning for Identity-Preserving Text-to-Video Generation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24962","citing_title":"Tempered Self-Similarity Alignment for Physically Plausible Video Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29095","citing_title":"HorizonRelight: Relighting Long-horizon Videos Consistently via Diffusion Transformers","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00499","citing_title":"OptiWorld: Optimal Control for Video World Generation under Physical Constraints","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23178","citing_title":"Composing People Together: Iterative Pose-Image Generation for Multi-Person Interaction Scenes","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23878","citing_title":"LaMo: Self-Supervised Latent Motion Priors for Physical Realism in Video Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14505","citing_title":"MusicInfuser: Making Video Diffusion Listen and Dance","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2601.10632","citing_title":"CoMoVi: Co-Generation of 3D Human Motions and Realistic Videos","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2601.16933","citing_title":"Reward-Forcing: Autoregressive Video Generation with Reward Feedback","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02283","citing_title":"Self-Forcing++: Towards Minute-Scale High-Quality Video Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14269","citing_title":"PhyMotion: Structured 3D Motion Reward for Physics-Grounded Human Video Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12587","citing_title":"TrackCraft3R: Repurposing Video Diffusion Transformers for Dense 3D Tracking","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13030","citing_title":"Motus: A Unified Latent Action World Model","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04515","citing_title":"From Priors to Perception: Grounding Video-LLMs in Physical Reality","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19636","citing_title":"CoInteract: Physically-Consistent Human-Object Interaction Video Synthesis via Spatially-Structured Co-Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05961","citing_title":"HumANDiff: Articulated Noise Diffusion for Motion-Consistent Human Video Generation","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LNS3Y4ZIRYLB3YVR3JNHYKA2K4","json":"https://pith.science/pith/LNS3Y4ZIRYLB3YVR3JNHYKA2K4.json","graph_json":"https://pith.science/api/pith-number/LNS3Y4ZIRYLB3YVR3JNHYKA2K4/graph.json","events_json":"https://pith.science/api/pith-number/LNS3Y4ZIRYLB3YVR3JNHYKA2K4/events.json","paper":"https://pith.science/paper/LNS3Y4ZI"},"agent_actions":{"view_html":"https://pith.science/pith/LNS3Y4ZIRYLB3YVR3JNHYKA2K4","download_json":"https://pith.science/pith/LNS3Y4ZIRYLB3YVR3JNHYKA2K4.json","view_paper":"https://pith.science/paper/LNS3Y4ZI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.02492&json=true","fetch_graph":"https://pith.science/api/pith-number/LNS3Y4ZIRYLB3YVR3JNHYKA2K4/graph.json","fetch_events":"https://pith.science/api/pith-number/LNS3Y4ZIRYLB3YVR3JNHYKA2K4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LNS3Y4ZIRYLB3YVR3JNHYKA2K4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LNS3Y4ZIRYLB3YVR3JNHYKA2K4/action/storage_attestation","attest_author":"https://pith.science/pith/LNS3Y4ZIRYLB3YVR3JNHYKA2K4/action/author_attestation","sign_citation":"https://pith.science/pith/LNS3Y4ZIRYLB3YVR3JNHYKA2K4/action/citation_signature","submit_replication":"https://pith.science/pith/LNS3Y4ZIRYLB3YVR3JNHYKA2K4/action/replication_record"}},"created_at":"2026-07-05T11:09:44.968697+00:00","updated_at":"2026-07-05T11:09:44.968697+00:00"}