{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RF4PBXB2NT3VIE2U3CLBNZLCLX","short_pith_number":"pith:RF4PBXB2","schema_version":"1.0","canonical_sha256":"8978f0dc3a6cf7541354d89616e5625dd9e27eecf894bbdb171611ea206a80f1","source":{"kind":"arxiv","id":"2407.07614","version":2},"attestation_state":"computed","paper":{"title":"MARS: Mixture of Auto-Regressive Models for Fine-grained Text-to-image Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fangxun Shu, Hao Jiang, Haoyuan Li, Leilei Gan, Lei Zhang, Mushui Liu, Siming Fu, Wanggui He, Wenyi Xiao, Xierui Wang, Yi Wang, Zhelun Yu, Ziwei Huang","submitted_at":"2024-07-10T12:52:49Z","abstract_excerpt":"Auto-regressive models have made significant progress in the realm of language generation, yet they do not perform on par with diffusion models in the domain of image synthesis. In this work, we introduce MARS, a novel framework for T2I generation that incorporates a specially designed Semantic Vision-Language Integration Expert (SemVIE). This innovative component integrates pre-trained LLMs by independently processing linguistic and visual information, freezing the textual component while fine-tuning the visual component. This methodology preserves the NLP capabilities of LLMs while imbuing t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.07614","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-10T12:52:49Z","cross_cats_sorted":[],"title_canon_sha256":"312e837b04fbbf43464e9b8f064dfaa1ab8775ce4a4bed514712dbac9596325d","abstract_canon_sha256":"94da94ed441d1d7b19635508277de00cbdb2963e71c4cbc40c8704576ddf2bec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:42:39.389924Z","signature_b64":"AsklKkMyvUQ29If/9kKxI2fxN+FwDkQp0Ju4wrHbiEEw9pPXAhNkBQoNGXWkJOcdYGpzjyNBH8op0TL8Rq89Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8978f0dc3a6cf7541354d89616e5625dd9e27eecf894bbdb171611ea206a80f1","last_reissued_at":"2026-07-05T08:42:39.389491Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:42:39.389491Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MARS: Mixture of Auto-Regressive Models for Fine-grained Text-to-image Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fangxun Shu, Hao Jiang, Haoyuan Li, Leilei Gan, Lei Zhang, Mushui Liu, Siming Fu, Wanggui He, Wenyi Xiao, Xierui Wang, Yi Wang, Zhelun Yu, Ziwei Huang","submitted_at":"2024-07-10T12:52:49Z","abstract_excerpt":"Auto-regressive models have made significant progress in the realm of language generation, yet they do not perform on par with diffusion models in the domain of image synthesis. In this work, we introduce MARS, a novel framework for T2I generation that incorporates a specially designed Semantic Vision-Language Integration Expert (SemVIE). This innovative component integrates pre-trained LLMs by independently processing linguistic and visual information, freezing the textual component while fine-tuning the visual component. This methodology preserves the NLP capabilities of LLMs while imbuing t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.07614","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.07614/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.07614","created_at":"2026-07-05T08:42:39.389548+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.07614v2","created_at":"2026-07-05T08:42:39.389548+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.07614","created_at":"2026-07-05T08:42:39.389548+00:00"},{"alias_kind":"pith_short_12","alias_value":"RF4PBXB2NT3V","created_at":"2026-07-05T08:42:39.389548+00:00"},{"alias_kind":"pith_short_16","alias_value":"RF4PBXB2NT3VIE2U","created_at":"2026-07-05T08:42:39.389548+00:00"},{"alias_kind":"pith_short_8","alias_value":"RF4PBXB2","created_at":"2026-07-05T08:42:39.389548+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08302","citing_title":"HACK++: Towards More Effective Head-Aware Key-Value Compression for Efficient Visual Autoregressive Modeling","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28615","citing_title":"Compositional Text-to-Image Generation Via Region-aware Bimodal Direct Preference Optimization","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04300","citing_title":"T2I-FactualBench: Benchmarking the Factuality of Text-to-Image Models with Knowledge-Intensive Concepts","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2411.04996","citing_title":"Mixture-of-Transformers: A Sparse and Scalable Architecture for Multi-Modal Foundation Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2410.24164","citing_title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RF4PBXB2NT3VIE2U3CLBNZLCLX","json":"https://pith.science/pith/RF4PBXB2NT3VIE2U3CLBNZLCLX.json","graph_json":"https://pith.science/api/pith-number/RF4PBXB2NT3VIE2U3CLBNZLCLX/graph.json","events_json":"https://pith.science/api/pith-number/RF4PBXB2NT3VIE2U3CLBNZLCLX/events.json","paper":"https://pith.science/paper/RF4PBXB2"},"agent_actions":{"view_html":"https://pith.science/pith/RF4PBXB2NT3VIE2U3CLBNZLCLX","download_json":"https://pith.science/pith/RF4PBXB2NT3VIE2U3CLBNZLCLX.json","view_paper":"https://pith.science/paper/RF4PBXB2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.07614&json=true","fetch_graph":"https://pith.science/api/pith-number/RF4PBXB2NT3VIE2U3CLBNZLCLX/graph.json","fetch_events":"https://pith.science/api/pith-number/RF4PBXB2NT3VIE2U3CLBNZLCLX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RF4PBXB2NT3VIE2U3CLBNZLCLX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RF4PBXB2NT3VIE2U3CLBNZLCLX/action/storage_attestation","attest_author":"https://pith.science/pith/RF4PBXB2NT3VIE2U3CLBNZLCLX/action/author_attestation","sign_citation":"https://pith.science/pith/RF4PBXB2NT3VIE2U3CLBNZLCLX/action/citation_signature","submit_replication":"https://pith.science/pith/RF4PBXB2NT3VIE2U3CLBNZLCLX/action/replication_record"}},"created_at":"2026-07-05T08:42:39.389548+00:00","updated_at":"2026-07-05T08:42:39.389548+00:00"}