{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RRT6ZOYAYG6T64OQ2VQYJKJMA5","short_pith_number":"pith:RRT6ZOYA","schema_version":"1.0","canonical_sha256":"8c67ecbb00c1bd3f71d0d56184a92c077738fadfd8735a404e848f1944300ec6","source":{"kind":"arxiv","id":"2305.18247","version":2},"attestation_state":"computed","paper":{"title":"TaleCrafter: Interactive Story Visualization with Multiple Characters","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haoxin Chen, Longyue Wang, Menghan Xia, Xiaodong Cun, Xintao Wang, Yingqing He, Ying Shan, Yong Zhang, Youxin Pang, Yuan Gong, Yujiu Yang","submitted_at":"2023-05-29T17:11:39Z","abstract_excerpt":"Accurate Story visualization requires several necessary elements, such as identity consistency across frames, the alignment between plain text and visual content, and a reasonable layout of objects in images. Most previous works endeavor to meet these requirements by fitting a text-to-image (T2I) model on a set of videos in the same style and with the same characters, e.g., the FlintstonesSV dataset. However, the learned T2I models typically struggle to adapt to new characters, scenes, and styles, and often lack the flexibility to revise the layout of the synthesized images. This paper propose"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.18247","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-05-29T17:11:39Z","cross_cats_sorted":[],"title_canon_sha256":"9f50bc9e092c511ae1befea8032703769a5c4a5fb89d470926f0ecbfd67c2be4","abstract_canon_sha256":"60965cc42e374dc95ff1a585e5f80f417a3d0bc2df11802417ce8d2c786c6100"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:15:33.342965Z","signature_b64":"PqvY0RyruTQddLKkBPiPnBeslikpBTg0QE35qHQ39IDk8/wNX3XqKFzY5righuf84dlW/hmedJ9N3BRXZD+2Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c67ecbb00c1bd3f71d0d56184a92c077738fadfd8735a404e848f1944300ec6","last_reissued_at":"2026-07-05T06:15:33.342509Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:15:33.342509Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TaleCrafter: Interactive Story Visualization with Multiple Characters","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haoxin Chen, Longyue Wang, Menghan Xia, Xiaodong Cun, Xintao Wang, Yingqing He, Ying Shan, Yong Zhang, Youxin Pang, Yuan Gong, Yujiu Yang","submitted_at":"2023-05-29T17:11:39Z","abstract_excerpt":"Accurate Story visualization requires several necessary elements, such as identity consistency across frames, the alignment between plain text and visual content, and a reasonable layout of objects in images. Most previous works endeavor to meet these requirements by fitting a text-to-image (T2I) model on a set of videos in the same style and with the same characters, e.g., the FlintstonesSV dataset. However, the learned T2I models typically struggle to adapt to new characters, scenes, and styles, and often lack the flexibility to revise the layout of the synthesized images. This paper propose"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18247","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.18247/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.18247","created_at":"2026-07-05T06:15:33.342566+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.18247v2","created_at":"2026-07-05T06:15:33.342566+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18247","created_at":"2026-07-05T06:15:33.342566+00:00"},{"alias_kind":"pith_short_12","alias_value":"RRT6ZOYAYG6T","created_at":"2026-07-05T06:15:33.342566+00:00"},{"alias_kind":"pith_short_16","alias_value":"RRT6ZOYAYG6T64OQ","created_at":"2026-07-05T06:15:33.342566+00:00"},{"alias_kind":"pith_short_8","alias_value":"RRT6ZOYA","created_at":"2026-07-05T06:15:33.342566+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.23690","citing_title":"SynMotion: Semantic-Visual Adaptation for Motion Customized Video Generation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28489","citing_title":"Video Generation Models as World Models: Efficient Paradigms, Architectures and Algorithms","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11927","citing_title":"RealDiffusion: Physics-informed Attention for Multi-character Storybook Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16541","citing_title":"BOOKAGENT: Orchestrating Safety-Aware Visual Narratives via Multi-Agent Cognitive Calibration","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RRT6ZOYAYG6T64OQ2VQYJKJMA5","json":"https://pith.science/pith/RRT6ZOYAYG6T64OQ2VQYJKJMA5.json","graph_json":"https://pith.science/api/pith-number/RRT6ZOYAYG6T64OQ2VQYJKJMA5/graph.json","events_json":"https://pith.science/api/pith-number/RRT6ZOYAYG6T64OQ2VQYJKJMA5/events.json","paper":"https://pith.science/paper/RRT6ZOYA"},"agent_actions":{"view_html":"https://pith.science/pith/RRT6ZOYAYG6T64OQ2VQYJKJMA5","download_json":"https://pith.science/pith/RRT6ZOYAYG6T64OQ2VQYJKJMA5.json","view_paper":"https://pith.science/paper/RRT6ZOYA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.18247&json=true","fetch_graph":"https://pith.science/api/pith-number/RRT6ZOYAYG6T64OQ2VQYJKJMA5/graph.json","fetch_events":"https://pith.science/api/pith-number/RRT6ZOYAYG6T64OQ2VQYJKJMA5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RRT6ZOYAYG6T64OQ2VQYJKJMA5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RRT6ZOYAYG6T64OQ2VQYJKJMA5/action/storage_attestation","attest_author":"https://pith.science/pith/RRT6ZOYAYG6T64OQ2VQYJKJMA5/action/author_attestation","sign_citation":"https://pith.science/pith/RRT6ZOYAYG6T64OQ2VQYJKJMA5/action/citation_signature","submit_replication":"https://pith.science/pith/RRT6ZOYAYG6T64OQ2VQYJKJMA5/action/replication_record"}},"created_at":"2026-07-05T06:15:33.342566+00:00","updated_at":"2026-07-05T06:15:33.342566+00:00"}