{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QWRBV77AK3CLBFJXF7CROKGVDP","short_pith_number":"pith:QWRBV77A","schema_version":"1.0","canonical_sha256":"85a21affe056c4b095372fc51728d51bdb38e293d81764b8315ec20f8ca5628e","source":{"kind":"arxiv","id":"2306.07257","version":1},"attestation_state":"computed","paper":{"title":"MovieFactory: Automatic Movie Creation from Text using Large Generative Models for Language and Images","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huan Yang, Huiguo He, Jianlong Fu, Jingkuan Song, Junchen Zhu, Lianli Gao, Wen-Huang Cheng, Wenjing Wang, Zixi Tuo","submitted_at":"2023-06-12T17:31:23Z","abstract_excerpt":"In this paper, we present MovieFactory, a powerful framework to generate cinematic-picture (3072$\\times$1280), film-style (multi-scene), and multi-modality (sounding) movies on the demand of natural languages. As the first fully automated movie generation model to the best of our knowledge, our approach empowers users to create captivating movies with smooth transitions using simple text inputs, surpassing existing methods that produce soundless videos limited to a single scene of modest quality. To facilitate this distinctive functionality, we leverage ChatGPT to expand user-provided text int"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.07257","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-06-12T17:31:23Z","cross_cats_sorted":[],"title_canon_sha256":"707d6309f55cc77fce2e716a3942b09b878bab6e8e9288d73e476620323eddc1","abstract_canon_sha256":"25ce81852f2ba77288faee8d23bcf55a6902aeaf3ac07530ed8dc28b4fb439d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:19:49.348660Z","signature_b64":"ViFV3QVo8iPvnuEqAgLTsrQ6ddjPT9q9hrWYashqj8Aiz9NbcbF3+nsmbkfl++7tLKw8YD5RENv12oBZ96AeDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85a21affe056c4b095372fc51728d51bdb38e293d81764b8315ec20f8ca5628e","last_reissued_at":"2026-07-05T06:19:49.348199Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:19:49.348199Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MovieFactory: Automatic Movie Creation from Text using Large Generative Models for Language and Images","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huan Yang, Huiguo He, Jianlong Fu, Jingkuan Song, Junchen Zhu, Lianli Gao, Wen-Huang Cheng, Wenjing Wang, Zixi Tuo","submitted_at":"2023-06-12T17:31:23Z","abstract_excerpt":"In this paper, we present MovieFactory, a powerful framework to generate cinematic-picture (3072$\\times$1280), film-style (multi-scene), and multi-modality (sounding) movies on the demand of natural languages. As the first fully automated movie generation model to the best of our knowledge, our approach empowers users to create captivating movies with smooth transitions using simple text inputs, surpassing existing methods that produce soundless videos limited to a single scene of modest quality. To facilitate this distinctive functionality, we leverage ChatGPT to expand user-provided text int"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.07257","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.07257/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.07257","created_at":"2026-07-05T06:19:49.348258+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.07257v1","created_at":"2026-07-05T06:19:49.348258+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.07257","created_at":"2026-07-05T06:19:49.348258+00:00"},{"alias_kind":"pith_short_12","alias_value":"QWRBV77AK3CL","created_at":"2026-07-05T06:19:49.348258+00:00"},{"alias_kind":"pith_short_16","alias_value":"QWRBV77AK3CLBFJX","created_at":"2026-07-05T06:19:49.348258+00:00"},{"alias_kind":"pith_short_8","alias_value":"QWRBV77A","created_at":"2026-07-05T06:19:49.348258+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2402.17177","citing_title":"Sora: A Review on Background, Technology, Limitations, and Opportunities of Large Vision Models","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2404.02101","citing_title":"CameraCtrl: Enabling Camera Control for Text-to-Video Generation","ref_index":64,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QWRBV77AK3CLBFJXF7CROKGVDP","json":"https://pith.science/pith/QWRBV77AK3CLBFJXF7CROKGVDP.json","graph_json":"https://pith.science/api/pith-number/QWRBV77AK3CLBFJXF7CROKGVDP/graph.json","events_json":"https://pith.science/api/pith-number/QWRBV77AK3CLBFJXF7CROKGVDP/events.json","paper":"https://pith.science/paper/QWRBV77A"},"agent_actions":{"view_html":"https://pith.science/pith/QWRBV77AK3CLBFJXF7CROKGVDP","download_json":"https://pith.science/pith/QWRBV77AK3CLBFJXF7CROKGVDP.json","view_paper":"https://pith.science/paper/QWRBV77A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.07257&json=true","fetch_graph":"https://pith.science/api/pith-number/QWRBV77AK3CLBFJXF7CROKGVDP/graph.json","fetch_events":"https://pith.science/api/pith-number/QWRBV77AK3CLBFJXF7CROKGVDP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QWRBV77AK3CLBFJXF7CROKGVDP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QWRBV77AK3CLBFJXF7CROKGVDP/action/storage_attestation","attest_author":"https://pith.science/pith/QWRBV77AK3CLBFJXF7CROKGVDP/action/author_attestation","sign_citation":"https://pith.science/pith/QWRBV77AK3CLBFJXF7CROKGVDP/action/citation_signature","submit_replication":"https://pith.science/pith/QWRBV77AK3CLBFJXF7CROKGVDP/action/replication_record"}},"created_at":"2026-07-05T06:19:49.348258+00:00","updated_at":"2026-07-05T06:19:49.348258+00:00"}