{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LVUKULFVS7DR7K26BV3TPMQMOD","short_pith_number":"pith:LVUKULFV","schema_version":"1.0","canonical_sha256":"5d68aa2cb597c71fab5e0d7737b20c70d016a276639f2e5adf4b5ba6070aa04e","source":{"kind":"arxiv","id":"2407.08683","version":2},"attestation_state":"computed","paper":{"title":"SEED-Story: Multimodal Long Story Generation with Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Shuai Yang, Yang Li, Yingcong Chen, Ying Shan, Yixiao Ge, Yukang Chen, Yuying Ge","submitted_at":"2024-07-11T17:21:03Z","abstract_excerpt":"With the remarkable advancements in image generation and open-form text generation, the creation of interleaved image-text content has become an increasingly intriguing field. Multimodal story generation, characterized by producing narrative texts and vivid images in an interleaved manner, has emerged as a valuable and practical task with broad applications. However, this task poses significant challenges, as it necessitates the comprehension of the complex interplay between texts and images, and the ability to generate long sequences of coherent, contextually relevant texts and visuals. In th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.08683","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-11T17:21:03Z","cross_cats_sorted":[],"title_canon_sha256":"7475856442c79fad40451c1eb3e1ba576acf32c61b1d25315461deacbfed84af","abstract_canon_sha256":"60b72838b8b4142537d03f6252556d79754a30849e7d244a133aa9f278e35d5e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:11.224841Z","signature_b64":"SfJvvkHLF+Dizjp1abA6NuTEOI420N8m4aPXA1C4DL0UH2tEMrYzI3evM2iMqbTET1KtDTBVONCDRqSteHPoAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d68aa2cb597c71fab5e0d7737b20c70d016a276639f2e5adf4b5ba6070aa04e","last_reissued_at":"2026-07-05T09:19:11.224268Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:11.224268Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SEED-Story: Multimodal Long Story Generation with Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Shuai Yang, Yang Li, Yingcong Chen, Ying Shan, Yixiao Ge, Yukang Chen, Yuying Ge","submitted_at":"2024-07-11T17:21:03Z","abstract_excerpt":"With the remarkable advancements in image generation and open-form text generation, the creation of interleaved image-text content has become an increasingly intriguing field. Multimodal story generation, characterized by producing narrative texts and vivid images in an interleaved manner, has emerged as a valuable and practical task with broad applications. However, this task poses significant challenges, as it necessitates the comprehension of the complex interplay between texts and images, and the ability to generate long sequences of coherent, contextually relevant texts and visuals. In th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.08683","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.08683/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.08683","created_at":"2026-07-05T09:19:11.224336+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.08683v2","created_at":"2026-07-05T09:19:11.224336+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.08683","created_at":"2026-07-05T09:19:11.224336+00:00"},{"alias_kind":"pith_short_12","alias_value":"LVUKULFVS7DR","created_at":"2026-07-05T09:19:11.224336+00:00"},{"alias_kind":"pith_short_16","alias_value":"LVUKULFVS7DR7K26","created_at":"2026-07-05T09:19:11.224336+00:00"},{"alias_kind":"pith_short_8","alias_value":"LVUKULFV","created_at":"2026-07-05T09:19:11.224336+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25079","citing_title":"FreeStory: Training-Free Character Consistency for Free-Form Visual Storytelling","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16819","citing_title":"Character-Centered Dialogue Generation from Scene-Level Prompts","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2511.15408","citing_title":"Chinese Short-Form Creative Content Generation via Explanation-Oriented Multi-Objective Optimization","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2410.10781","citing_title":"When Attention Sink Emerges in Language Models: An Empirical View","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2501.04001","citing_title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22622","citing_title":"LongLive: Real-time Interactive Long Video Generation","ref_index":107,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LVUKULFVS7DR7K26BV3TPMQMOD","json":"https://pith.science/pith/LVUKULFVS7DR7K26BV3TPMQMOD.json","graph_json":"https://pith.science/api/pith-number/LVUKULFVS7DR7K26BV3TPMQMOD/graph.json","events_json":"https://pith.science/api/pith-number/LVUKULFVS7DR7K26BV3TPMQMOD/events.json","paper":"https://pith.science/paper/LVUKULFV"},"agent_actions":{"view_html":"https://pith.science/pith/LVUKULFVS7DR7K26BV3TPMQMOD","download_json":"https://pith.science/pith/LVUKULFVS7DR7K26BV3TPMQMOD.json","view_paper":"https://pith.science/paper/LVUKULFV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.08683&json=true","fetch_graph":"https://pith.science/api/pith-number/LVUKULFVS7DR7K26BV3TPMQMOD/graph.json","fetch_events":"https://pith.science/api/pith-number/LVUKULFVS7DR7K26BV3TPMQMOD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LVUKULFVS7DR7K26BV3TPMQMOD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LVUKULFVS7DR7K26BV3TPMQMOD/action/storage_attestation","attest_author":"https://pith.science/pith/LVUKULFVS7DR7K26BV3TPMQMOD/action/author_attestation","sign_citation":"https://pith.science/pith/LVUKULFVS7DR7K26BV3TPMQMOD/action/citation_signature","submit_replication":"https://pith.science/pith/LVUKULFVS7DR7K26BV3TPMQMOD/action/replication_record"}},"created_at":"2026-07-05T09:19:11.224336+00:00","updated_at":"2026-07-05T09:19:11.224336+00:00"}