{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XDRK7DZY4IHZBVFHEQ6TYOQVYU","short_pith_number":"pith:XDRK7DZY","schema_version":"1.0","canonical_sha256":"b8e2af8f38e20f90d4a7243d3c3a15c51fd7557c49f6a39d2cc5afa00212421a","source":{"kind":"arxiv","id":"2402.01566","version":1},"attestation_state":"computed","paper":{"title":"Boximator: Generating Rich and Controllable Motions for Video Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Guoqiang Wei, Hang Li, Jiawei Wang, Jiaxin Zou, Liping Yuan, Yan Zeng, Yuchen Zhang","submitted_at":"2024-02-02T16:59:48Z","abstract_excerpt":"Generating rich and controllable motion is a pivotal challenge in video synthesis. We propose Boximator, a new approach for fine-grained motion control. Boximator introduces two constraint types: hard box and soft box. Users select objects in the conditional frame using hard boxes and then use either type of boxes to roughly or rigorously define the object's position, shape, or motion path in future frames. Boximator functions as a plug-in for existing video diffusion models. Its training process preserves the base model's knowledge by freezing the original weights and training only the contro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.01566","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-02T16:59:48Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"751465163eb8ffd03dfc58b337e74afc73181834e6491d61907d468b8261eb78","abstract_canon_sha256":"59cc6cbcaf2c767b01696ee41764c999bbb33d983e1cd745ed0907c7664f2dd4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:40:45.936135Z","signature_b64":"B6JzlXBtveVhlDlDjm+NrjWDtWImICRD+hxsUl8S6Ub6cyfdJqkrR8Esh/yMr8nHvoW7flsjtanuZnOnGUyIAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8e2af8f38e20f90d4a7243d3c3a15c51fd7557c49f6a39d2cc5afa00212421a","last_reissued_at":"2026-07-05T07:40:45.935697Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:40:45.935697Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Boximator: Generating Rich and Controllable Motions for Video Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Guoqiang Wei, Hang Li, Jiawei Wang, Jiaxin Zou, Liping Yuan, Yan Zeng, Yuchen Zhang","submitted_at":"2024-02-02T16:59:48Z","abstract_excerpt":"Generating rich and controllable motion is a pivotal challenge in video synthesis. We propose Boximator, a new approach for fine-grained motion control. Boximator introduces two constraint types: hard box and soft box. Users select objects in the conditional frame using hard boxes and then use either type of boxes to roughly or rigorously define the object's position, shape, or motion path in future frames. Boximator functions as a plug-in for existing video diffusion models. Its training process preserves the base model's knowledge by freezing the original weights and training only the contro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.01566","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.01566/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.01566","created_at":"2026-07-05T07:40:45.935752+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.01566v1","created_at":"2026-07-05T07:40:45.935752+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.01566","created_at":"2026-07-05T07:40:45.935752+00:00"},{"alias_kind":"pith_short_12","alias_value":"XDRK7DZY4IHZ","created_at":"2026-07-05T07:40:45.935752+00:00"},{"alias_kind":"pith_short_16","alias_value":"XDRK7DZY4IHZBVFH","created_at":"2026-07-05T07:40:45.935752+00:00"},{"alias_kind":"pith_short_8","alias_value":"XDRK7DZY","created_at":"2026-07-05T07:40:45.935752+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08770","citing_title":"LongE2V: Long-Horizon Event-based Video Reconstruction, Prediction, and Frame Interpolation with Video Diffusion Models","ref_index":86,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19495","citing_title":"LooseControlVideo: Directorial Video Control using Spatial Blocking","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02517","citing_title":"WorldDirector: Building Controllable World Simulators with Persistent Dynamic Memory","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31529","citing_title":"SVI-Bench: A Dynamic Microworld for Strategic Video Intelligence","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05162","citing_title":"Controllable Dynamic 3D Shape Generation via 3D Trajectories and Text","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14988","citing_title":"Compositional Video Generation via Inference-Time Guidance","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31529","citing_title":"SVI-Bench: A Dynamic Microworld for Strategic Video Intelligence","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20961","citing_title":"Preserve, Reveal, Expand: Faithful 4D Video Editing with Region-Aware Conditioning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03305","citing_title":"HVG-3D: Bridging Real and Simulation Domains for 3D-Conditional Hand-Object Interaction Video Synthesis","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28169","citing_title":"PhyCo: Learning Controllable Physical Priors for Generative Motion","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07966","citing_title":"Lighting-grounded Video Generation with Renderer-based Agent Reasoning","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07348","citing_title":"MoRight: Motion Control Done Right","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":217,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XDRK7DZY4IHZBVFHEQ6TYOQVYU","json":"https://pith.science/pith/XDRK7DZY4IHZBVFHEQ6TYOQVYU.json","graph_json":"https://pith.science/api/pith-number/XDRK7DZY4IHZBVFHEQ6TYOQVYU/graph.json","events_json":"https://pith.science/api/pith-number/XDRK7DZY4IHZBVFHEQ6TYOQVYU/events.json","paper":"https://pith.science/paper/XDRK7DZY"},"agent_actions":{"view_html":"https://pith.science/pith/XDRK7DZY4IHZBVFHEQ6TYOQVYU","download_json":"https://pith.science/pith/XDRK7DZY4IHZBVFHEQ6TYOQVYU.json","view_paper":"https://pith.science/paper/XDRK7DZY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.01566&json=true","fetch_graph":"https://pith.science/api/pith-number/XDRK7DZY4IHZBVFHEQ6TYOQVYU/graph.json","fetch_events":"https://pith.science/api/pith-number/XDRK7DZY4IHZBVFHEQ6TYOQVYU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XDRK7DZY4IHZBVFHEQ6TYOQVYU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XDRK7DZY4IHZBVFHEQ6TYOQVYU/action/storage_attestation","attest_author":"https://pith.science/pith/XDRK7DZY4IHZBVFHEQ6TYOQVYU/action/author_attestation","sign_citation":"https://pith.science/pith/XDRK7DZY4IHZBVFHEQ6TYOQVYU/action/citation_signature","submit_replication":"https://pith.science/pith/XDRK7DZY4IHZBVFHEQ6TYOQVYU/action/replication_record"}},"created_at":"2026-07-05T07:40:45.935752+00:00","updated_at":"2026-07-05T07:40:45.935752+00:00"}