{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MRSCII7AIERPU7AFIVMLRQQQSN","short_pith_number":"pith:MRSCII7A","schema_version":"1.0","canonical_sha256":"64642423e04122fa7c054558b8c2109344299b1e0141106303f12ced025f6624","source":{"kind":"arxiv","id":"2412.00127","version":2},"attestation_state":"computed","paper":{"title":"Orthus: Autoregressive Interleaved Image-Text Generation with Modality-Specific Heads","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chang Liu, Jiachun Jin, Jian Jia, Peng Jiang, Quan Chen, Siqi Kou, Ye Ma, Zhihong Liu, Zhijie Deng","submitted_at":"2024-11-28T13:00:38Z","abstract_excerpt":"We introduce Orthus, an autoregressive (AR) transformer that excels in generating images given textual prompts, answering questions based on visual inputs, and even crafting lengthy image-text interleaved contents. Unlike prior arts on unified multimodal modeling, Orthus simultaneously copes with discrete text tokens and continuous image features under the AR modeling principle. The continuous treatment of visual signals minimizes the information loss for both image understanding and generation while the fully AR formulation renders the characterization of the correlation between modalities st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.00127","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-28T13:00:38Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"475cd7a01508f3833ec7973a5a8e38f6560b2c0c56d2738ee59ff1b4ee6d67f5","abstract_canon_sha256":"edd5407f4d74e5b8b5016003aecb418a33a2361d76f327de10d736de983921a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:49:49.411580Z","signature_b64":"ypGacpuzCCj7vO4XfHm3/fcc6gOPrGMssPokNMHNrfS7MHsUTnQVkmBk4+XVInttP3TfVDrH0XrGfgGLg17nAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"64642423e04122fa7c054558b8c2109344299b1e0141106303f12ced025f6624","last_reissued_at":"2026-07-05T10:49:49.411051Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:49:49.411051Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Orthus: Autoregressive Interleaved Image-Text Generation with Modality-Specific Heads","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chang Liu, Jiachun Jin, Jian Jia, Peng Jiang, Quan Chen, Siqi Kou, Ye Ma, Zhihong Liu, Zhijie Deng","submitted_at":"2024-11-28T13:00:38Z","abstract_excerpt":"We introduce Orthus, an autoregressive (AR) transformer that excels in generating images given textual prompts, answering questions based on visual inputs, and even crafting lengthy image-text interleaved contents. Unlike prior arts on unified multimodal modeling, Orthus simultaneously copes with discrete text tokens and continuous image features under the AR modeling principle. The continuous treatment of visual signals minimizes the information loss for both image understanding and generation while the fully AR formulation renders the characterization of the correlation between modalities st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.00127","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.00127/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.00127","created_at":"2026-07-05T10:49:49.411119+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.00127v2","created_at":"2026-07-05T10:49:49.411119+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.00127","created_at":"2026-07-05T10:49:49.411119+00:00"},{"alias_kind":"pith_short_12","alias_value":"MRSCII7AIERP","created_at":"2026-07-05T10:49:49.411119+00:00"},{"alias_kind":"pith_short_16","alias_value":"MRSCII7AIERPU7AF","created_at":"2026-07-05T10:49:49.411119+00:00"},{"alias_kind":"pith_short_8","alias_value":"MRSCII7A","created_at":"2026-07-05T10:49:49.411119+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13289","citing_title":"HYDRA-X: Native Unified Multimodal Models with Holistic Visual Tokenizers","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01022","citing_title":"ProductWebGen: Benchmarking Multimodal Product Webpage Generation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16933","citing_title":"LLaDA-V: Large Language Diffusion Models with Visual Instruction Tuning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07584","citing_title":"LongCat-Image Technical Report","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07265","citing_title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15564","citing_title":"Show-o2: Improved Native Unified Multimodal Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06966","citing_title":"MAR-GRPO: Stabilized GRPO for AR-diffusion Hybrid Image Generation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05005","citing_title":"EduIllustrate: Towards Scalable Automated Generation Of Multimodal Educational Content","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MRSCII7AIERPU7AFIVMLRQQQSN","json":"https://pith.science/pith/MRSCII7AIERPU7AFIVMLRQQQSN.json","graph_json":"https://pith.science/api/pith-number/MRSCII7AIERPU7AFIVMLRQQQSN/graph.json","events_json":"https://pith.science/api/pith-number/MRSCII7AIERPU7AFIVMLRQQQSN/events.json","paper":"https://pith.science/paper/MRSCII7A"},"agent_actions":{"view_html":"https://pith.science/pith/MRSCII7AIERPU7AFIVMLRQQQSN","download_json":"https://pith.science/pith/MRSCII7AIERPU7AFIVMLRQQQSN.json","view_paper":"https://pith.science/paper/MRSCII7A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.00127&json=true","fetch_graph":"https://pith.science/api/pith-number/MRSCII7AIERPU7AFIVMLRQQQSN/graph.json","fetch_events":"https://pith.science/api/pith-number/MRSCII7AIERPU7AFIVMLRQQQSN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MRSCII7AIERPU7AFIVMLRQQQSN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MRSCII7AIERPU7AFIVMLRQQQSN/action/storage_attestation","attest_author":"https://pith.science/pith/MRSCII7AIERPU7AFIVMLRQQQSN/action/author_attestation","sign_citation":"https://pith.science/pith/MRSCII7AIERPU7AFIVMLRQQQSN/action/citation_signature","submit_replication":"https://pith.science/pith/MRSCII7AIERPU7AFIVMLRQQQSN/action/replication_record"}},"created_at":"2026-07-05T10:49:49.411119+00:00","updated_at":"2026-07-05T10:49:49.411119+00:00"}