{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5FLTEO3ZX4KZ5IDM4VRL3MEW56","short_pith_number":"pith:5FLTEO3Z","schema_version":"1.0","canonical_sha256":"e957323b79bf159ea06ce562bdb096efa3b5d197bc75499261c2172bdffe0b31","source":{"kind":"arxiv","id":"2502.17425","version":1},"attestation_state":"computed","paper":{"title":"Introducing Visual Perception Token into Multimodal Large Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Runpeng Yu, Xinchao Wang, Xinyin Ma","submitted_at":"2025-02-24T18:56:12Z","abstract_excerpt":"To utilize visual information, Multimodal Large Language Model (MLLM) relies on the perception process of its vision encoder. The completeness and accuracy of visual perception significantly influence the precision of spatial reasoning, fine-grained understanding, and other tasks. However, MLLM still lacks the autonomous capability to control its own visual perception processes, for example, selectively reviewing specific regions of an image or focusing on information related to specific object categories. In this work, we propose the concept of Visual Perception Token, aiming to empower MLLM "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.17425","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-24T18:56:12Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e3a70a66e0cce735e36919810f5ce6afd2f3dba9121c5b0582cdc1125fe1da0b","abstract_canon_sha256":"809b237c450bbd36856d4c751c35844716aee8fdb7b7cc618c360c8c4dce6b85"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:10.015957Z","signature_b64":"45KJNpRM2Pyra4vUP9CMF7HedxY22quWTvUObgN96Ynfc4ZEreTD5K69ZaG65rj4bRxX8tYJD36iFdF8eKLOCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e957323b79bf159ea06ce562bdb096efa3b5d197bc75499261c2172bdffe0b31","last_reissued_at":"2026-07-05T10:19:10.015486Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:10.015486Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Introducing Visual Perception Token into Multimodal Large Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Runpeng Yu, Xinchao Wang, Xinyin Ma","submitted_at":"2025-02-24T18:56:12Z","abstract_excerpt":"To utilize visual information, Multimodal Large Language Model (MLLM) relies on the perception process of its vision encoder. The completeness and accuracy of visual perception significantly influence the precision of spatial reasoning, fine-grained understanding, and other tasks. However, MLLM still lacks the autonomous capability to control its own visual perception processes, for example, selectively reviewing specific regions of an image or focusing on information related to specific object categories. In this work, we propose the concept of Visual Perception Token, aiming to empower MLLM "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.17425","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.17425/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.17425","created_at":"2026-07-05T10:19:10.015537+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.17425v1","created_at":"2026-07-05T10:19:10.015537+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.17425","created_at":"2026-07-05T10:19:10.015537+00:00"},{"alias_kind":"pith_short_12","alias_value":"5FLTEO3ZX4KZ","created_at":"2026-07-05T10:19:10.015537+00:00"},{"alias_kind":"pith_short_16","alias_value":"5FLTEO3ZX4KZ5IDM","created_at":"2026-07-05T10:19:10.015537+00:00"},{"alias_kind":"pith_short_8","alias_value":"5FLTEO3Z","created_at":"2026-07-05T10:19:10.015537+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08434","citing_title":"DeltaV: Thinking with Visual State Updates in Unified Large Multimodal Models","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":139,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17888","citing_title":"MathVis-Fine: Aligning Visual Supervision with Necessity via Progressive Dependency-Guided Training for Multimodal Mathematical Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30168","citing_title":"Latent Noise Mask for Reducing Visual Redundancy in Multimodal Large Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27959","citing_title":"ROVER: Routing Object-Centric Visual Evidence for Grounded Multi-Image Reasoning","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18600","citing_title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15951","citing_title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18600","citing_title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02870","citing_title":"Token Warping Helps MLLMs Look from Nearby Viewpoints","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10500","citing_title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10500","citing_title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12896","citing_title":"Don't Show Pixels, Show Cues: Unlocking Visual Tool Reasoning in Language Models via Perception Programs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10500","citing_title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5FLTEO3ZX4KZ5IDM4VRL3MEW56","json":"https://pith.science/pith/5FLTEO3ZX4KZ5IDM4VRL3MEW56.json","graph_json":"https://pith.science/api/pith-number/5FLTEO3ZX4KZ5IDM4VRL3MEW56/graph.json","events_json":"https://pith.science/api/pith-number/5FLTEO3ZX4KZ5IDM4VRL3MEW56/events.json","paper":"https://pith.science/paper/5FLTEO3Z"},"agent_actions":{"view_html":"https://pith.science/pith/5FLTEO3ZX4KZ5IDM4VRL3MEW56","download_json":"https://pith.science/pith/5FLTEO3ZX4KZ5IDM4VRL3MEW56.json","view_paper":"https://pith.science/paper/5FLTEO3Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.17425&json=true","fetch_graph":"https://pith.science/api/pith-number/5FLTEO3ZX4KZ5IDM4VRL3MEW56/graph.json","fetch_events":"https://pith.science/api/pith-number/5FLTEO3ZX4KZ5IDM4VRL3MEW56/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5FLTEO3ZX4KZ5IDM4VRL3MEW56/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5FLTEO3ZX4KZ5IDM4VRL3MEW56/action/storage_attestation","attest_author":"https://pith.science/pith/5FLTEO3ZX4KZ5IDM4VRL3MEW56/action/author_attestation","sign_citation":"https://pith.science/pith/5FLTEO3ZX4KZ5IDM4VRL3MEW56/action/citation_signature","submit_replication":"https://pith.science/pith/5FLTEO3ZX4KZ5IDM4VRL3MEW56/action/replication_record"}},"created_at":"2026-07-05T10:19:10.015537+00:00","updated_at":"2026-07-05T10:19:10.015537+00:00"}