{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3CTVALZJI4O45RR2TFY63TG4RY","short_pith_number":"pith:3CTVALZJ","schema_version":"1.0","canonical_sha256":"d8a7502f29471dcec63a9971edccdc8e15c5d07a3a4872f3693f6b97abe8beb8","source":{"kind":"arxiv","id":"2303.16839","version":3},"attestation_state":"computed","paper":{"title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Abhijit Ogale, AJ Piergiovanni, Andrew Dai, Anelia Angelova, Ben Caine, Claire Cui, Dahun Kim, Luowei Zhou, Weicheng Kuo, Wei Li, Xiyang Luo, Zhifeng Chen","submitted_at":"2023-03-29T16:42:30Z","abstract_excerpt":"The development of language models have moved from encoder-decoder to decoder-only designs. In addition, we observe that the two most popular multimodal tasks, the generative and contrastive tasks, are nontrivial to accommodate in one architecture, and further need adaptations for downstream tasks. We propose a novel paradigm of training with a decoder-only model for multimodal tasks, which is surprisingly effective in jointly learning of these disparate vision-language tasks. This is done with a simple model, called MaMMUT. It consists of a single vision encoder and a text decoder, and is abl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.16839","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-03-29T16:42:30Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"806a32304bc0c48552c92d6d9fc01eff56c8b6fdff70968428be3cf8a0a2fe02","abstract_canon_sha256":"3919696c311721cfa1ba4c36f6af624475dd05a5f421b106516be83fa273ead8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:39:33.679170Z","signature_b64":"g9ZCtS42XWxGYqeNQ3JfJ8yrvDIWJ9odH8jdYketsTnep3pl0dtguP9MGv7vVkvkFRly4AloJU7cBDyOfeP2Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d8a7502f29471dcec63a9971edccdc8e15c5d07a3a4872f3693f6b97abe8beb8","last_reissued_at":"2026-07-05T06:39:33.678665Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:39:33.678665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Abhijit Ogale, AJ Piergiovanni, Andrew Dai, Anelia Angelova, Ben Caine, Claire Cui, Dahun Kim, Luowei Zhou, Weicheng Kuo, Wei Li, Xiyang Luo, Zhifeng Chen","submitted_at":"2023-03-29T16:42:30Z","abstract_excerpt":"The development of language models have moved from encoder-decoder to decoder-only designs. In addition, we observe that the two most popular multimodal tasks, the generative and contrastive tasks, are nontrivial to accommodate in one architecture, and further need adaptations for downstream tasks. We propose a novel paradigm of training with a decoder-only model for multimodal tasks, which is surprisingly effective in jointly learning of these disparate vision-language tasks. This is done with a simple model, called MaMMUT. It consists of a single vision encoder and a text decoder, and is abl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.16839","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.16839/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.16839","created_at":"2026-07-05T06:39:33.678729+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.16839v3","created_at":"2026-07-05T06:39:33.678729+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.16839","created_at":"2026-07-05T06:39:33.678729+00:00"},{"alias_kind":"pith_short_12","alias_value":"3CTVALZJI4O4","created_at":"2026-07-05T06:39:33.678729+00:00"},{"alias_kind":"pith_short_16","alias_value":"3CTVALZJI4O45RR2","created_at":"2026-07-05T06:39:33.678729+00:00"},{"alias_kind":"pith_short_8","alias_value":"3CTVALZJ","created_at":"2026-07-05T06:39:33.678729+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.20623","citing_title":"RSRCC: A Remote Sensing Regional Change Comprehension Benchmark Constructed via Retrieval-Augmented Best-of-N Ranking","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3CTVALZJI4O45RR2TFY63TG4RY","json":"https://pith.science/pith/3CTVALZJI4O45RR2TFY63TG4RY.json","graph_json":"https://pith.science/api/pith-number/3CTVALZJI4O45RR2TFY63TG4RY/graph.json","events_json":"https://pith.science/api/pith-number/3CTVALZJI4O45RR2TFY63TG4RY/events.json","paper":"https://pith.science/paper/3CTVALZJ"},"agent_actions":{"view_html":"https://pith.science/pith/3CTVALZJI4O45RR2TFY63TG4RY","download_json":"https://pith.science/pith/3CTVALZJI4O45RR2TFY63TG4RY.json","view_paper":"https://pith.science/paper/3CTVALZJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.16839&json=true","fetch_graph":"https://pith.science/api/pith-number/3CTVALZJI4O45RR2TFY63TG4RY/graph.json","fetch_events":"https://pith.science/api/pith-number/3CTVALZJI4O45RR2TFY63TG4RY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3CTVALZJI4O45RR2TFY63TG4RY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3CTVALZJI4O45RR2TFY63TG4RY/action/storage_attestation","attest_author":"https://pith.science/pith/3CTVALZJI4O45RR2TFY63TG4RY/action/author_attestation","sign_citation":"https://pith.science/pith/3CTVALZJI4O45RR2TFY63TG4RY/action/citation_signature","submit_replication":"https://pith.science/pith/3CTVALZJI4O45RR2TFY63TG4RY/action/replication_record"}},"created_at":"2026-07-05T06:39:33.678729+00:00","updated_at":"2026-07-05T06:39:33.678729+00:00"}