{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YX7JQWNREVSQEBLY46DTDHNMJP","short_pith_number":"pith:YX7JQWNR","schema_version":"1.0","canonical_sha256":"c5fe9859b12565020578e787319dac4be7dc0be049c24f368baf91754deeb6ab","source":{"kind":"arxiv","id":"2410.10802","version":1},"attestation_state":"computed","paper":{"title":"Boosting Camera Motion Control for Video Diffusion Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Andrew Gilbert, Armin Mustafa, Chun-Hao Paul Huang, Duygu Ceylan, Soon Yau Cheong","submitted_at":"2024-10-14T17:58:07Z","abstract_excerpt":"Recent advancements in diffusion models have significantly enhanced the quality of video generation. However, fine-grained control over camera pose remains a challenge. While U-Net-based models have shown promising results for camera control, transformer-based diffusion models (DiT)-the preferred architecture for large-scale video generation - suffer from severe degradation in camera motion accuracy. In this paper, we investigate the underlying causes of this issue and propose solutions tailored to DiT architectures. Our study reveals that camera control performance depends heavily on the choi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.10802","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-14T17:58:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5a6d34574bed7b6feb90b4cd01039836d49165b61fb89bcbaba891993eaf7ce7","abstract_canon_sha256":"76fc1203f89f2df8fe6ee62834841990429ee49d69a21cf47189d16fc3438f33"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:20.983417Z","signature_b64":"8ckLH+oj0FCmJb3085V4oIdA9/9EpKccZ0ytEplUanAsQy6VKzETdJYN+n81uwONU1ClsbstwqMlgOw1aTWaCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c5fe9859b12565020578e787319dac4be7dc0be049c24f368baf91754deeb6ab","last_reissued_at":"2026-07-05T09:20:20.982943Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:20.982943Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Boosting Camera Motion Control for Video Diffusion Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Andrew Gilbert, Armin Mustafa, Chun-Hao Paul Huang, Duygu Ceylan, Soon Yau Cheong","submitted_at":"2024-10-14T17:58:07Z","abstract_excerpt":"Recent advancements in diffusion models have significantly enhanced the quality of video generation. However, fine-grained control over camera pose remains a challenge. While U-Net-based models have shown promising results for camera control, transformer-based diffusion models (DiT)-the preferred architecture for large-scale video generation - suffer from severe degradation in camera motion accuracy. In this paper, we investigate the underlying causes of this issue and propose solutions tailored to DiT architectures. Our study reveals that camera control performance depends heavily on the choi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.10802","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.10802/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.10802","created_at":"2026-07-05T09:20:20.983002+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.10802v1","created_at":"2026-07-05T09:20:20.983002+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.10802","created_at":"2026-07-05T09:20:20.983002+00:00"},{"alias_kind":"pith_short_12","alias_value":"YX7JQWNREVSQ","created_at":"2026-07-05T09:20:20.983002+00:00"},{"alias_kind":"pith_short_16","alias_value":"YX7JQWNREVSQEBLY","created_at":"2026-07-05T09:20:20.983002+00:00"},{"alias_kind":"pith_short_8","alias_value":"YX7JQWNR","created_at":"2026-07-05T09:20:20.983002+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.01099","citing_title":"Geometry-aware 4D Video Generation for Robot Manipulation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2404.02101","citing_title":"CameraCtrl: Enabling Camera Control for Text-to-Video Generation","ref_index":108,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YX7JQWNREVSQEBLY46DTDHNMJP","json":"https://pith.science/pith/YX7JQWNREVSQEBLY46DTDHNMJP.json","graph_json":"https://pith.science/api/pith-number/YX7JQWNREVSQEBLY46DTDHNMJP/graph.json","events_json":"https://pith.science/api/pith-number/YX7JQWNREVSQEBLY46DTDHNMJP/events.json","paper":"https://pith.science/paper/YX7JQWNR"},"agent_actions":{"view_html":"https://pith.science/pith/YX7JQWNREVSQEBLY46DTDHNMJP","download_json":"https://pith.science/pith/YX7JQWNREVSQEBLY46DTDHNMJP.json","view_paper":"https://pith.science/paper/YX7JQWNR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.10802&json=true","fetch_graph":"https://pith.science/api/pith-number/YX7JQWNREVSQEBLY46DTDHNMJP/graph.json","fetch_events":"https://pith.science/api/pith-number/YX7JQWNREVSQEBLY46DTDHNMJP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YX7JQWNREVSQEBLY46DTDHNMJP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YX7JQWNREVSQEBLY46DTDHNMJP/action/storage_attestation","attest_author":"https://pith.science/pith/YX7JQWNREVSQEBLY46DTDHNMJP/action/author_attestation","sign_citation":"https://pith.science/pith/YX7JQWNREVSQEBLY46DTDHNMJP/action/citation_signature","submit_replication":"https://pith.science/pith/YX7JQWNREVSQEBLY46DTDHNMJP/action/replication_record"}},"created_at":"2026-07-05T09:20:20.983002+00:00","updated_at":"2026-07-05T09:20:20.983002+00:00"}