{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QWKS556CJPBL6CRAP7PEFVMCZH","short_pith_number":"pith:QWKS556C","schema_version":"1.0","canonical_sha256":"85952ef7c24bc2bf0a207fde42d582c9d91f5c5b8501ee9baff93a49bcdf7526","source":{"kind":"arxiv","id":"2409.00558","version":1},"attestation_state":"computed","paper":{"title":"Compositional 3D-aware Video Generation with LLM Director","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anni Tang, Hanxin Zhu, Jiang Bian, Junliang Guo, Tianyu He, Zhibo Chen","submitted_at":"2024-08-31T23:07:22Z","abstract_excerpt":"Significant progress has been made in text-to-video generation through the use of powerful generative models and large-scale internet data. However, substantial challenges remain in precisely controlling individual concepts within the generated video, such as the motion and appearance of specific characters and the movement of viewpoints. In this work, we propose a novel paradigm that generates each concept in 3D representation separately and then composes them with priors from Large Language Models (LLM) and 2D diffusion models. Specifically, given an input textual prompt, our scheme consists"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.00558","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-31T23:07:22Z","cross_cats_sorted":[],"title_canon_sha256":"5e771c8f862730f0a3b9faf42a31711ec5a69e69308802e2152f4943e2601eba","abstract_canon_sha256":"7f53ea33af933f5d498aca33bde5659ec09e50e226c199070de66d4c2466ce85"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:30.770559Z","signature_b64":"OUNsgnFepKtjOsYUf+z7BlDVlmrp7SWpk2K9DtMam5tx201TqmPf7t0V/pAjUhdL+l64UaRznQReAJt6jS5dCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85952ef7c24bc2bf0a207fde42d582c9d91f5c5b8501ee9baff93a49bcdf7526","last_reissued_at":"2026-07-05T09:12:30.770094Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:30.770094Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Compositional 3D-aware Video Generation with LLM Director","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anni Tang, Hanxin Zhu, Jiang Bian, Junliang Guo, Tianyu He, Zhibo Chen","submitted_at":"2024-08-31T23:07:22Z","abstract_excerpt":"Significant progress has been made in text-to-video generation through the use of powerful generative models and large-scale internet data. However, substantial challenges remain in precisely controlling individual concepts within the generated video, such as the motion and appearance of specific characters and the movement of viewpoints. In this work, we propose a novel paradigm that generates each concept in 3D representation separately and then composes them with priors from Large Language Models (LLM) and 2D diffusion models. Specifically, given an input textual prompt, our scheme consists"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.00558","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.00558/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.00558","created_at":"2026-07-05T09:12:30.770149+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.00558v1","created_at":"2026-07-05T09:12:30.770149+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.00558","created_at":"2026-07-05T09:12:30.770149+00:00"},{"alias_kind":"pith_short_12","alias_value":"QWKS556CJPBL","created_at":"2026-07-05T09:12:30.770149+00:00"},{"alias_kind":"pith_short_16","alias_value":"QWKS556CJPBL6CRA","created_at":"2026-07-05T09:12:30.770149+00:00"},{"alias_kind":"pith_short_8","alias_value":"QWKS556C","created_at":"2026-07-05T09:12:30.770149+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.10928","citing_title":"Generative Physical AI in Vision: A Survey","ref_index":228,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QWKS556CJPBL6CRAP7PEFVMCZH","json":"https://pith.science/pith/QWKS556CJPBL6CRAP7PEFVMCZH.json","graph_json":"https://pith.science/api/pith-number/QWKS556CJPBL6CRAP7PEFVMCZH/graph.json","events_json":"https://pith.science/api/pith-number/QWKS556CJPBL6CRAP7PEFVMCZH/events.json","paper":"https://pith.science/paper/QWKS556C"},"agent_actions":{"view_html":"https://pith.science/pith/QWKS556CJPBL6CRAP7PEFVMCZH","download_json":"https://pith.science/pith/QWKS556CJPBL6CRAP7PEFVMCZH.json","view_paper":"https://pith.science/paper/QWKS556C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.00558&json=true","fetch_graph":"https://pith.science/api/pith-number/QWKS556CJPBL6CRAP7PEFVMCZH/graph.json","fetch_events":"https://pith.science/api/pith-number/QWKS556CJPBL6CRAP7PEFVMCZH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QWKS556CJPBL6CRAP7PEFVMCZH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QWKS556CJPBL6CRAP7PEFVMCZH/action/storage_attestation","attest_author":"https://pith.science/pith/QWKS556CJPBL6CRAP7PEFVMCZH/action/author_attestation","sign_citation":"https://pith.science/pith/QWKS556CJPBL6CRAP7PEFVMCZH/action/citation_signature","submit_replication":"https://pith.science/pith/QWKS556CJPBL6CRAP7PEFVMCZH/action/replication_record"}},"created_at":"2026-07-05T09:12:30.770149+00:00","updated_at":"2026-07-05T09:12:30.770149+00:00"}