{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KAW6N2U3G7F3WMG4M5QQY7TEEN","short_pith_number":"pith:KAW6N2U3","schema_version":"1.0","canonical_sha256":"502de6ea9b37cbbb30dc67610c7e64234a4065eb62ca59a72054eef7e5f60150","source":{"kind":"arxiv","id":"2410.13861","version":2},"attestation_state":"computed","paper":{"title":"PUMA: Empowering Unified MLLM with Multi-granular Visual Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengqi Duan, Hao Li, Hao Tian, Hongsheng Li, Jifeng Dai, Kun Wang, Rongyao Fang, Rui Zhao, Xihui Liu, Xingyu Zeng","submitted_at":"2024-10-17T17:59:57Z","abstract_excerpt":"Recent advancements in multimodal foundation models have yielded significant progress in vision-language understanding. Initial attempts have also explored the potential of multimodal large language models (MLLMs) for visual content generation. However, existing works have insufficiently addressed the varying granularity demands of different image generation tasks within a unified MLLM paradigm - from the diversity required in text-to-image generation to the precise controllability needed in image manipulation. In this work, we propose PUMA, emPowering Unified MLLM with Multi-grAnular visual g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.13861","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-17T17:59:57Z","cross_cats_sorted":[],"title_canon_sha256":"76af0f3d0ebc61798e38f86791ee14f54526592aa3f2db6eccb04cc0d6fac329","abstract_canon_sha256":"ca3e5e588fdc9a4a2f5b78de510d85d117c02997767ec0410e3b2cd9a6588321"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:23:15.585929Z","signature_b64":"5Da2HDDkvr3El2SAaC2nUjP6oF74uel+eQql49pAjvLNJAiBGkP6GVzWqWYRR9qUQSP4raBLAfSaWPZEuVfGDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"502de6ea9b37cbbb30dc67610c7e64234a4065eb62ca59a72054eef7e5f60150","last_reissued_at":"2026-07-05T09:23:15.585519Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:23:15.585519Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PUMA: Empowering Unified MLLM with Multi-granular Visual Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengqi Duan, Hao Li, Hao Tian, Hongsheng Li, Jifeng Dai, Kun Wang, Rongyao Fang, Rui Zhao, Xihui Liu, Xingyu Zeng","submitted_at":"2024-10-17T17:59:57Z","abstract_excerpt":"Recent advancements in multimodal foundation models have yielded significant progress in vision-language understanding. Initial attempts have also explored the potential of multimodal large language models (MLLMs) for visual content generation. However, existing works have insufficiently addressed the varying granularity demands of different image generation tasks within a unified MLLM paradigm - from the diversity required in text-to-image generation to the precise controllability needed in image manipulation. In this work, we propose PUMA, emPowering Unified MLLM with Multi-grAnular visual g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.13861","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.13861/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.13861","created_at":"2026-07-05T09:23:15.585576+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.13861v2","created_at":"2026-07-05T09:23:15.585576+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.13861","created_at":"2026-07-05T09:23:15.585576+00:00"},{"alias_kind":"pith_short_12","alias_value":"KAW6N2U3G7F3","created_at":"2026-07-05T09:23:15.585576+00:00"},{"alias_kind":"pith_short_16","alias_value":"KAW6N2U3G7F3WMG4","created_at":"2026-07-05T09:23:15.585576+00:00"},{"alias_kind":"pith_short_8","alias_value":"KAW6N2U3","created_at":"2026-07-05T09:23:15.585576+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.27902","citing_title":"One Patch Is Enough: Reinforcement-Optimized Visual Token Grounding for MLLM-Based Scene Text Spotting","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KAW6N2U3G7F3WMG4M5QQY7TEEN","json":"https://pith.science/pith/KAW6N2U3G7F3WMG4M5QQY7TEEN.json","graph_json":"https://pith.science/api/pith-number/KAW6N2U3G7F3WMG4M5QQY7TEEN/graph.json","events_json":"https://pith.science/api/pith-number/KAW6N2U3G7F3WMG4M5QQY7TEEN/events.json","paper":"https://pith.science/paper/KAW6N2U3"},"agent_actions":{"view_html":"https://pith.science/pith/KAW6N2U3G7F3WMG4M5QQY7TEEN","download_json":"https://pith.science/pith/KAW6N2U3G7F3WMG4M5QQY7TEEN.json","view_paper":"https://pith.science/paper/KAW6N2U3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.13861&json=true","fetch_graph":"https://pith.science/api/pith-number/KAW6N2U3G7F3WMG4M5QQY7TEEN/graph.json","fetch_events":"https://pith.science/api/pith-number/KAW6N2U3G7F3WMG4M5QQY7TEEN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KAW6N2U3G7F3WMG4M5QQY7TEEN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KAW6N2U3G7F3WMG4M5QQY7TEEN/action/storage_attestation","attest_author":"https://pith.science/pith/KAW6N2U3G7F3WMG4M5QQY7TEEN/action/author_attestation","sign_citation":"https://pith.science/pith/KAW6N2U3G7F3WMG4M5QQY7TEEN/action/citation_signature","submit_replication":"https://pith.science/pith/KAW6N2U3G7F3WMG4M5QQY7TEEN/action/replication_record"}},"created_at":"2026-07-05T09:23:15.585576+00:00","updated_at":"2026-07-05T09:23:15.585576+00:00"}