{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KRQ3K66BZDRELWL6OS6NS37NLN","short_pith_number":"pith:KRQ3K66B","schema_version":"1.0","canonical_sha256":"5461b57bc1c8e245d97e74bcd96fed5b61b85e2bdf949671deac76ea0e8018a2","source":{"kind":"arxiv","id":"2311.16500","version":4},"attestation_state":"computed","paper":{"title":"LLMGA: Multimodal Large Language Model based Generation Assistant","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Xia, Jiaya Jia, Shiyin Wang, Yingfan Tao, Yitong Wang","submitted_at":"2023-11-27T13:37:26Z","abstract_excerpt":"In this paper, we introduce a Multimodal Large Language Model-based Generation Assistant (LLMGA), leveraging the vast reservoir of knowledge and proficiency in reasoning, comprehension, and response inherent in Large Language Models (LLMs) to assist users in image generation and editing. Diverging from existing approaches where Multimodal Large Language Models (MLLMs) generate fixed-size embeddings to control Stable Diffusion (SD), our LLMGA provides a detailed language generation prompt for precise control over SD. This not only augments LLM context understanding but also reduces noise in gen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.16500","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-11-27T13:37:26Z","cross_cats_sorted":[],"title_canon_sha256":"33a540be95c62b015e247d47ca353863350be2174f1964cfe5113103ae71b76b","abstract_canon_sha256":"ac32bf4465d58d78ba8e1292b1028c9f9099510ba4899f3a742f5060132de423"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:58.880490Z","signature_b64":"+GZuuEqlau0/gs092lUgAs36GYf4eFWECfF/nJKKAMJcMMLY+KToiU86GiC+ONt+XBY+gd+wrrjS9xzHbzJ0DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5461b57bc1c8e245d97e74bcd96fed5b61b85e2bdf949671deac76ea0e8018a2","last_reissued_at":"2026-07-05T08:48:58.880071Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:58.880071Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMGA: Multimodal Large Language Model based Generation Assistant","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Xia, Jiaya Jia, Shiyin Wang, Yingfan Tao, Yitong Wang","submitted_at":"2023-11-27T13:37:26Z","abstract_excerpt":"In this paper, we introduce a Multimodal Large Language Model-based Generation Assistant (LLMGA), leveraging the vast reservoir of knowledge and proficiency in reasoning, comprehension, and response inherent in Large Language Models (LLMs) to assist users in image generation and editing. Diverging from existing approaches where Multimodal Large Language Models (MLLMs) generate fixed-size embeddings to control Stable Diffusion (SD), our LLMGA provides a detailed language generation prompt for precise control over SD. This not only augments LLM context understanding but also reduces noise in gen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.16500","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.16500/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.16500","created_at":"2026-07-05T08:48:58.880128+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.16500v4","created_at":"2026-07-05T08:48:58.880128+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.16500","created_at":"2026-07-05T08:48:58.880128+00:00"},{"alias_kind":"pith_short_12","alias_value":"KRQ3K66BZDRE","created_at":"2026-07-05T08:48:58.880128+00:00"},{"alias_kind":"pith_short_16","alias_value":"KRQ3K66BZDRELWL6","created_at":"2026-07-05T08:48:58.880128+00:00"},{"alias_kind":"pith_short_8","alias_value":"KRQ3K66B","created_at":"2026-07-05T08:48:58.880128+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20764","citing_title":"One Image is All You Need: Agentic One-Shot Image Generation via Text-Based World Models for Long-Tail Spatial Perception","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2403.18814","citing_title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KRQ3K66BZDRELWL6OS6NS37NLN","json":"https://pith.science/pith/KRQ3K66BZDRELWL6OS6NS37NLN.json","graph_json":"https://pith.science/api/pith-number/KRQ3K66BZDRELWL6OS6NS37NLN/graph.json","events_json":"https://pith.science/api/pith-number/KRQ3K66BZDRELWL6OS6NS37NLN/events.json","paper":"https://pith.science/paper/KRQ3K66B"},"agent_actions":{"view_html":"https://pith.science/pith/KRQ3K66BZDRELWL6OS6NS37NLN","download_json":"https://pith.science/pith/KRQ3K66BZDRELWL6OS6NS37NLN.json","view_paper":"https://pith.science/paper/KRQ3K66B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.16500&json=true","fetch_graph":"https://pith.science/api/pith-number/KRQ3K66BZDRELWL6OS6NS37NLN/graph.json","fetch_events":"https://pith.science/api/pith-number/KRQ3K66BZDRELWL6OS6NS37NLN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KRQ3K66BZDRELWL6OS6NS37NLN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KRQ3K66BZDRELWL6OS6NS37NLN/action/storage_attestation","attest_author":"https://pith.science/pith/KRQ3K66BZDRELWL6OS6NS37NLN/action/author_attestation","sign_citation":"https://pith.science/pith/KRQ3K66BZDRELWL6OS6NS37NLN/action/citation_signature","submit_replication":"https://pith.science/pith/KRQ3K66BZDRELWL6OS6NS37NLN/action/replication_record"}},"created_at":"2026-07-05T08:48:58.880128+00:00","updated_at":"2026-07-05T08:48:58.880128+00:00"}