{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BSG2OSBZMMAPPFFUKZGPFVBMWC","short_pith_number":"pith:BSG2OSBZ","schema_version":"1.0","canonical_sha256":"0c8da748396300f794b4564cf2d42cb09dfd8e0602b0124550f9cd0d12d94dee","source":{"kind":"arxiv","id":"2503.13026","version":2},"attestation_state":"computed","paper":{"title":"HiMTok: Learning Hierarchical Mask Tokens for Image Segmentation with Large Multimodal Model","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changxu Cheng, Lingfeng Wang, Senda Chen, Tao Wang, Wuyue Zhao","submitted_at":"2025-03-17T10:29:08Z","abstract_excerpt":"The remarkable performance of large multimodal models (LMMs) has attracted significant interest from the image segmentation community. To align with the next-token-prediction paradigm, current LMM-driven segmentation methods either use object boundary points to represent masks or introduce special segmentation tokens, whose hidden states are decoded by a segmentation model requiring the original image as input. However, these approaches often suffer from inadequate mask representation and complex architectures, limiting the potential of LMMs. In this work, we propose the Hierarchical Mask Toke"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.13026","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-17T10:29:08Z","cross_cats_sorted":[],"title_canon_sha256":"cf7b429a714994f383177ccf6fb7632a3208bcdc6eac7e5091016c287799c373","abstract_canon_sha256":"b6e6dde2dabe8b199ebfcd58f81dcb229cf723440cc548c39cef017ef9492d34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:38:05.598108Z","signature_b64":"0y3j1KEaMc7iUQ5iAGf2YoFzmHPs+jV+q8nH5cOB82XYAZzN2B4dcawE3TA5lJXyuNYy749r06VPGWzGPv1UBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c8da748396300f794b4564cf2d42cb09dfd8e0602b0124550f9cd0d12d94dee","last_reissued_at":"2026-07-05T11:38:05.597609Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:38:05.597609Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HiMTok: Learning Hierarchical Mask Tokens for Image Segmentation with Large Multimodal Model","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changxu Cheng, Lingfeng Wang, Senda Chen, Tao Wang, Wuyue Zhao","submitted_at":"2025-03-17T10:29:08Z","abstract_excerpt":"The remarkable performance of large multimodal models (LMMs) has attracted significant interest from the image segmentation community. To align with the next-token-prediction paradigm, current LMM-driven segmentation methods either use object boundary points to represent masks or introduce special segmentation tokens, whose hidden states are decoded by a segmentation model requiring the original image as input. However, these approaches often suffer from inadequate mask representation and complex architectures, limiting the potential of LMMs. In this work, we propose the Hierarchical Mask Toke"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.13026","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.13026/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.13026","created_at":"2026-07-05T11:38:05.597667+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.13026v2","created_at":"2026-07-05T11:38:05.597667+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.13026","created_at":"2026-07-05T11:38:05.597667+00:00"},{"alias_kind":"pith_short_12","alias_value":"BSG2OSBZMMAP","created_at":"2026-07-05T11:38:05.597667+00:00"},{"alias_kind":"pith_short_16","alias_value":"BSG2OSBZMMAPPFFU","created_at":"2026-07-05T11:38:05.597667+00:00"},{"alias_kind":"pith_short_8","alias_value":"BSG2OSBZ","created_at":"2026-07-05T11:38:05.597667+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":112,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11789","citing_title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","ref_index":169,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BSG2OSBZMMAPPFFUKZGPFVBMWC","json":"https://pith.science/pith/BSG2OSBZMMAPPFFUKZGPFVBMWC.json","graph_json":"https://pith.science/api/pith-number/BSG2OSBZMMAPPFFUKZGPFVBMWC/graph.json","events_json":"https://pith.science/api/pith-number/BSG2OSBZMMAPPFFUKZGPFVBMWC/events.json","paper":"https://pith.science/paper/BSG2OSBZ"},"agent_actions":{"view_html":"https://pith.science/pith/BSG2OSBZMMAPPFFUKZGPFVBMWC","download_json":"https://pith.science/pith/BSG2OSBZMMAPPFFUKZGPFVBMWC.json","view_paper":"https://pith.science/paper/BSG2OSBZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.13026&json=true","fetch_graph":"https://pith.science/api/pith-number/BSG2OSBZMMAPPFFUKZGPFVBMWC/graph.json","fetch_events":"https://pith.science/api/pith-number/BSG2OSBZMMAPPFFUKZGPFVBMWC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BSG2OSBZMMAPPFFUKZGPFVBMWC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BSG2OSBZMMAPPFFUKZGPFVBMWC/action/storage_attestation","attest_author":"https://pith.science/pith/BSG2OSBZMMAPPFFUKZGPFVBMWC/action/author_attestation","sign_citation":"https://pith.science/pith/BSG2OSBZMMAPPFFUKZGPFVBMWC/action/citation_signature","submit_replication":"https://pith.science/pith/BSG2OSBZMMAPPFFUKZGPFVBMWC/action/replication_record"}},"created_at":"2026-07-05T11:38:05.597667+00:00","updated_at":"2026-07-05T11:38:05.597667+00:00"}