{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GYSIE3EBJ23M2C6GLRYFQ2YNKV","short_pith_number":"pith:GYSIE3EB","schema_version":"1.0","canonical_sha256":"3624826c814eb6cd0bc65c70586b0d5562cdaef77be5f1b018b165bc61124eb5","source":{"kind":"arxiv","id":"2404.13013","version":1},"attestation_state":"computed","paper":{"title":"Groma: Localized Visual Tokenization for Grounding Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chuofan Ma, Jiannan Wu, Xiaojuan Qi, Yi Jiang, Zehuan Yuan","submitted_at":"2024-04-19T17:22:51Z","abstract_excerpt":"We introduce Groma, a Multimodal Large Language Model (MLLM) with grounded and fine-grained visual perception ability. Beyond holistic image understanding, Groma is adept at region-level tasks such as region captioning and visual grounding. Such capabilities are built upon a localized visual tokenization mechanism, where an image input is decomposed into regions of interest and subsequently encoded into region tokens. By integrating region tokens into user instructions and model responses, we seamlessly enable Groma to understand user-specified region inputs and ground its textual output to im"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.13013","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-19T17:22:51Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"1b84b59db40e3f6b162dd00bc54c74618642ecc6c53d7af388f3f6f6feb5fdf7","abstract_canon_sha256":"56c436649851f2bb6ed662d4fa7cee173d0583039ba7ea4c60ff29cd0c4f1f85"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:10:01.957965Z","signature_b64":"Wf8j2EPRtXLR8JVwCWNnNzWqJKU37enhrcy0UcAlHyCl6hc8axn6LBbV+/UmYRtbYVjMGGh5Msd2zpigTdNgBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3624826c814eb6cd0bc65c70586b0d5562cdaef77be5f1b018b165bc61124eb5","last_reissued_at":"2026-07-05T08:10:01.957583Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:10:01.957583Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Groma: Localized Visual Tokenization for Grounding Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chuofan Ma, Jiannan Wu, Xiaojuan Qi, Yi Jiang, Zehuan Yuan","submitted_at":"2024-04-19T17:22:51Z","abstract_excerpt":"We introduce Groma, a Multimodal Large Language Model (MLLM) with grounded and fine-grained visual perception ability. Beyond holistic image understanding, Groma is adept at region-level tasks such as region captioning and visual grounding. Such capabilities are built upon a localized visual tokenization mechanism, where an image input is decomposed into regions of interest and subsequently encoded into region tokens. By integrating region tokens into user instructions and model responses, we seamlessly enable Groma to understand user-specified region inputs and ground its textual output to im"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.13013","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.13013/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.13013","created_at":"2026-07-05T08:10:01.957640+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.13013v1","created_at":"2026-07-05T08:10:01.957640+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.13013","created_at":"2026-07-05T08:10:01.957640+00:00"},{"alias_kind":"pith_short_12","alias_value":"GYSIE3EBJ23M","created_at":"2026-07-05T08:10:01.957640+00:00"},{"alias_kind":"pith_short_16","alias_value":"GYSIE3EBJ23M2C6G","created_at":"2026-07-05T08:10:01.957640+00:00"},{"alias_kind":"pith_short_8","alias_value":"GYSIE3EB","created_at":"2026-07-05T08:10:01.957640+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.10554","citing_title":"Grounding Everything in Tokens for Multimodal Large Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00270","citing_title":"OmniSch: A Multimodal PCB Schematic Benchmark For Structured Diagram Visual Reasoning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2406.06525","citing_title":"Autoregressive Model Beats Diffusion: Llama for Scalable Image Generation","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GYSIE3EBJ23M2C6GLRYFQ2YNKV","json":"https://pith.science/pith/GYSIE3EBJ23M2C6GLRYFQ2YNKV.json","graph_json":"https://pith.science/api/pith-number/GYSIE3EBJ23M2C6GLRYFQ2YNKV/graph.json","events_json":"https://pith.science/api/pith-number/GYSIE3EBJ23M2C6GLRYFQ2YNKV/events.json","paper":"https://pith.science/paper/GYSIE3EB"},"agent_actions":{"view_html":"https://pith.science/pith/GYSIE3EBJ23M2C6GLRYFQ2YNKV","download_json":"https://pith.science/pith/GYSIE3EBJ23M2C6GLRYFQ2YNKV.json","view_paper":"https://pith.science/paper/GYSIE3EB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.13013&json=true","fetch_graph":"https://pith.science/api/pith-number/GYSIE3EBJ23M2C6GLRYFQ2YNKV/graph.json","fetch_events":"https://pith.science/api/pith-number/GYSIE3EBJ23M2C6GLRYFQ2YNKV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GYSIE3EBJ23M2C6GLRYFQ2YNKV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GYSIE3EBJ23M2C6GLRYFQ2YNKV/action/storage_attestation","attest_author":"https://pith.science/pith/GYSIE3EBJ23M2C6GLRYFQ2YNKV/action/author_attestation","sign_citation":"https://pith.science/pith/GYSIE3EBJ23M2C6GLRYFQ2YNKV/action/citation_signature","submit_replication":"https://pith.science/pith/GYSIE3EBJ23M2C6GLRYFQ2YNKV/action/replication_record"}},"created_at":"2026-07-05T08:10:01.957640+00:00","updated_at":"2026-07-05T08:10:01.957640+00:00"}