{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZYQSOVQUWJAOWE5PVZ455EDHSA","short_pith_number":"pith:ZYQSOVQU","schema_version":"1.0","canonical_sha256":"ce21275614b240eb13afae79de9067900a3c9036959d48d8643350b02b433b30","source":{"kind":"arxiv","id":"2507.18009","version":1},"attestation_state":"computed","paper":{"title":"GRR-CoCa: Leveraging LLM Mechanisms in Multimodal Model Architectures","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Canling Chen, Christina Gomez, Jake R. Patock, Kevin McCoy, Lorenzo Luzi, Nicole Catherine Lewis","submitted_at":"2025-07-24T00:54:31Z","abstract_excerpt":"State-of-the-art (SOTA) image and text generation models are multimodal models that have many similarities to large language models (LLMs). Despite achieving strong performances, leading foundational multimodal model architectures frequently lag behind the architectural sophistication of contemporary LLMs. We propose GRR-CoCa, an improved SOTA Contrastive Captioner (CoCa) model that incorporates Gaussian error gated linear units, root mean squared normalization, and rotary positional embedding into the textual decoders and the vision transformer (ViT) encoder. Each architectural modification h"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.18009","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-07-24T00:54:31Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"43830386e37d2390e5793b3890b987354cc0e8e72321e8b093891eb5af29a108","abstract_canon_sha256":"f7538d9fc1ae2bb1f762198222b811f43a17ed7351f610d77ce692897df06e1e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:42:39.309088Z","signature_b64":"Dr0WVsNzcx6/LyhH5/UGgO2I4O6LRTX2N1n8kj9aC/yMBqZVn7LeHaFuYHK/ANPAhj2Yl1n2+UOLu6lmId9nCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce21275614b240eb13afae79de9067900a3c9036959d48d8643350b02b433b30","last_reissued_at":"2026-07-05T11:42:39.308603Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:42:39.308603Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GRR-CoCa: Leveraging LLM Mechanisms in Multimodal Model Architectures","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Canling Chen, Christina Gomez, Jake R. Patock, Kevin McCoy, Lorenzo Luzi, Nicole Catherine Lewis","submitted_at":"2025-07-24T00:54:31Z","abstract_excerpt":"State-of-the-art (SOTA) image and text generation models are multimodal models that have many similarities to large language models (LLMs). Despite achieving strong performances, leading foundational multimodal model architectures frequently lag behind the architectural sophistication of contemporary LLMs. We propose GRR-CoCa, an improved SOTA Contrastive Captioner (CoCa) model that incorporates Gaussian error gated linear units, root mean squared normalization, and rotary positional embedding into the textual decoders and the vision transformer (ViT) encoder. Each architectural modification h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.18009","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.18009/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.18009","created_at":"2026-07-05T11:42:39.308662+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.18009v1","created_at":"2026-07-05T11:42:39.308662+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.18009","created_at":"2026-07-05T11:42:39.308662+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZYQSOVQUWJAO","created_at":"2026-07-05T11:42:39.308662+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZYQSOVQUWJAOWE5P","created_at":"2026-07-05T11:42:39.308662+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZYQSOVQU","created_at":"2026-07-05T11:42:39.308662+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZYQSOVQUWJAOWE5PVZ455EDHSA","json":"https://pith.science/pith/ZYQSOVQUWJAOWE5PVZ455EDHSA.json","graph_json":"https://pith.science/api/pith-number/ZYQSOVQUWJAOWE5PVZ455EDHSA/graph.json","events_json":"https://pith.science/api/pith-number/ZYQSOVQUWJAOWE5PVZ455EDHSA/events.json","paper":"https://pith.science/paper/ZYQSOVQU"},"agent_actions":{"view_html":"https://pith.science/pith/ZYQSOVQUWJAOWE5PVZ455EDHSA","download_json":"https://pith.science/pith/ZYQSOVQUWJAOWE5PVZ455EDHSA.json","view_paper":"https://pith.science/paper/ZYQSOVQU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.18009&json=true","fetch_graph":"https://pith.science/api/pith-number/ZYQSOVQUWJAOWE5PVZ455EDHSA/graph.json","fetch_events":"https://pith.science/api/pith-number/ZYQSOVQUWJAOWE5PVZ455EDHSA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZYQSOVQUWJAOWE5PVZ455EDHSA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZYQSOVQUWJAOWE5PVZ455EDHSA/action/storage_attestation","attest_author":"https://pith.science/pith/ZYQSOVQUWJAOWE5PVZ455EDHSA/action/author_attestation","sign_citation":"https://pith.science/pith/ZYQSOVQUWJAOWE5PVZ455EDHSA/action/citation_signature","submit_replication":"https://pith.science/pith/ZYQSOVQUWJAOWE5PVZ455EDHSA/action/replication_record"}},"created_at":"2026-07-05T11:42:39.308662+00:00","updated_at":"2026-07-05T11:42:39.308662+00:00"}