{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5EPDMA4QU6BYPW4QPB2Y2WJDUI","short_pith_number":"pith:5EPDMA4Q","schema_version":"1.0","canonical_sha256":"e91e360390a78387db9078758d5923a231e7c452826a2205f04569932ae3892b","source":{"kind":"arxiv","id":"2505.21660","version":2},"attestation_state":"computed","paper":{"title":"PreGenie: An Agentic Framework for High-quality Visual Presentation Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Fan Zhang, Haoyu Chen, Sirui Chen, Xiaojie Xu, Xinli Xu, Ying-Cong Chen","submitted_at":"2025-05-27T18:36:19Z","abstract_excerpt":"Visual presentations are vital for effective communication. Early attempts to automate their creation using deep learning often faced issues such as poorly organized layouts, inaccurate text summarization, and a lack of image understanding, leading to mismatched visuals and text. These limitations restrict their application in formal contexts like business and scientific research. To address these challenges, we propose PreGenie, an agentic and modular framework powered by multimodal large language models (MLLMs) for generating high-quality visual presentations.\n  PreGenie is built on the Slid"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.21660","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-27T18:36:19Z","cross_cats_sorted":[],"title_canon_sha256":"d84100ab4e0f9f372570e9c8f018307cb580f6cb5a5d74d43d84ea17f7fa8dad","abstract_canon_sha256":"60d796689482499065eadabb3024809ec225ba41f0e69172c898fa2f54506ae1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:02:02.021681Z","signature_b64":"sp0lQnHSSLEBstpPYcmC35c3f2ySMl+Q/spcZhRYa7/jDJbUDX0mExxl/XrmpaoHdNKFY8HxDvavGc6BEvBeDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e91e360390a78387db9078758d5923a231e7c452826a2205f04569932ae3892b","last_reissued_at":"2026-07-05T12:02:02.021154Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:02:02.021154Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PreGenie: An Agentic Framework for High-quality Visual Presentation Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Fan Zhang, Haoyu Chen, Sirui Chen, Xiaojie Xu, Xinli Xu, Ying-Cong Chen","submitted_at":"2025-05-27T18:36:19Z","abstract_excerpt":"Visual presentations are vital for effective communication. Early attempts to automate their creation using deep learning often faced issues such as poorly organized layouts, inaccurate text summarization, and a lack of image understanding, leading to mismatched visuals and text. These limitations restrict their application in formal contexts like business and scientific research. To address these challenges, we propose PreGenie, an agentic and modular framework powered by multimodal large language models (MLLMs) for generating high-quality visual presentations.\n  PreGenie is built on the Slid"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.21660","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.21660/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.21660","created_at":"2026-07-05T12:02:02.021217+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.21660v2","created_at":"2026-07-05T12:02:02.021217+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.21660","created_at":"2026-07-05T12:02:02.021217+00:00"},{"alias_kind":"pith_short_12","alias_value":"5EPDMA4QU6BY","created_at":"2026-07-05T12:02:02.021217+00:00"},{"alias_kind":"pith_short_16","alias_value":"5EPDMA4QU6BYPW4Q","created_at":"2026-07-05T12:02:02.021217+00:00"},{"alias_kind":"pith_short_8","alias_value":"5EPDMA4Q","created_at":"2026-07-05T12:02:02.021217+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00407","citing_title":"Personalization as Inverse Planning: Learning Latent Design Intents for Agentic Slide Generation via Structural Denoising","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15202","citing_title":"DeepSlide: From Artifacts to Presentation Delivery","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11363","citing_title":"PresentAgent-2: Towards Generalist Multimodal Presentation Agents","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11969","citing_title":"Narrative-Driven Paper-to-Slide Generation via ArcDeck","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5EPDMA4QU6BYPW4QPB2Y2WJDUI","json":"https://pith.science/pith/5EPDMA4QU6BYPW4QPB2Y2WJDUI.json","graph_json":"https://pith.science/api/pith-number/5EPDMA4QU6BYPW4QPB2Y2WJDUI/graph.json","events_json":"https://pith.science/api/pith-number/5EPDMA4QU6BYPW4QPB2Y2WJDUI/events.json","paper":"https://pith.science/paper/5EPDMA4Q"},"agent_actions":{"view_html":"https://pith.science/pith/5EPDMA4QU6BYPW4QPB2Y2WJDUI","download_json":"https://pith.science/pith/5EPDMA4QU6BYPW4QPB2Y2WJDUI.json","view_paper":"https://pith.science/paper/5EPDMA4Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.21660&json=true","fetch_graph":"https://pith.science/api/pith-number/5EPDMA4QU6BYPW4QPB2Y2WJDUI/graph.json","fetch_events":"https://pith.science/api/pith-number/5EPDMA4QU6BYPW4QPB2Y2WJDUI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5EPDMA4QU6BYPW4QPB2Y2WJDUI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5EPDMA4QU6BYPW4QPB2Y2WJDUI/action/storage_attestation","attest_author":"https://pith.science/pith/5EPDMA4QU6BYPW4QPB2Y2WJDUI/action/author_attestation","sign_citation":"https://pith.science/pith/5EPDMA4QU6BYPW4QPB2Y2WJDUI/action/citation_signature","submit_replication":"https://pith.science/pith/5EPDMA4QU6BYPW4QPB2Y2WJDUI/action/replication_record"}},"created_at":"2026-07-05T12:02:02.021217+00:00","updated_at":"2026-07-05T12:02:02.021217+00:00"}