{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O7KZVJB2V37MX5EHPZFN47MW2M","short_pith_number":"pith:O7KZVJB2","schema_version":"1.0","canonical_sha256":"77d59aa43aaefecbf4877e4ade7d96d307f87740ad0a448ff9f8f92513750c25","source":{"kind":"arxiv","id":"2405.20339","version":1},"attestation_state":"computed","paper":{"title":"Visual Perception by Large Language Model's Weights","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feipeng Ma, Fengyun Rao, Guangting Wang, Hongwei Xue, Mike Zheng Shou, Shilin Yan, Siying Wu, Xiaoyan Sun, Yizhou Zhou, Yueyi Zhang","submitted_at":"2024-05-30T17:59:47Z","abstract_excerpt":"Existing Multimodal Large Language Models (MLLMs) follow the paradigm that perceives visual information by aligning visual features with the input space of Large Language Models (LLMs), and concatenating visual tokens with text tokens to form a unified sequence input for LLMs. These methods demonstrate promising results on various vision-language tasks but are limited by the high computational effort due to the extended input sequence resulting from the involvement of visual tokens. In this paper, instead of input space alignment, we propose a novel parameter space alignment paradigm that repr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.20339","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-30T17:59:47Z","cross_cats_sorted":[],"title_canon_sha256":"8b3baf24c9e092313f2c1621da6669b79ea110046d3255734d128ee840489f34","abstract_canon_sha256":"af727b53dacc6123493dbf374fbd37fbaffd46a2297848bd4c426494624300fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:24.554228Z","signature_b64":"pKdo+/d3HHLirgf9wTTDOJKeS7gOP9mxNVupLct3+L8q0ZqkUHXmVHKtIeiDHBKVmfXX9ETYYCIPAQteq9O/Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77d59aa43aaefecbf4877e4ade7d96d307f87740ad0a448ff9f8f92513750c25","last_reissued_at":"2026-07-05T08:25:24.553704Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:24.553704Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Perception by Large Language Model's Weights","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feipeng Ma, Fengyun Rao, Guangting Wang, Hongwei Xue, Mike Zheng Shou, Shilin Yan, Siying Wu, Xiaoyan Sun, Yizhou Zhou, Yueyi Zhang","submitted_at":"2024-05-30T17:59:47Z","abstract_excerpt":"Existing Multimodal Large Language Models (MLLMs) follow the paradigm that perceives visual information by aligning visual features with the input space of Large Language Models (LLMs), and concatenating visual tokens with text tokens to form a unified sequence input for LLMs. These methods demonstrate promising results on various vision-language tasks but are limited by the high computational effort due to the extended input sequence resulting from the involvement of visual tokens. In this paper, instead of input space alignment, we propose a novel parameter space alignment paradigm that repr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20339","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.20339/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.20339","created_at":"2026-07-05T08:25:24.553764+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.20339v1","created_at":"2026-07-05T08:25:24.553764+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20339","created_at":"2026-07-05T08:25:24.553764+00:00"},{"alias_kind":"pith_short_12","alias_value":"O7KZVJB2V37M","created_at":"2026-07-05T08:25:24.553764+00:00"},{"alias_kind":"pith_short_16","alias_value":"O7KZVJB2V37MX5EH","created_at":"2026-07-05T08:25:24.553764+00:00"},{"alias_kind":"pith_short_8","alias_value":"O7KZVJB2","created_at":"2026-07-05T08:25:24.553764+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.04326","citing_title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01833","citing_title":"Language-Pretraining-Induced Bias: A Strong Foundation for General Vision Tasks","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O7KZVJB2V37MX5EHPZFN47MW2M","json":"https://pith.science/pith/O7KZVJB2V37MX5EHPZFN47MW2M.json","graph_json":"https://pith.science/api/pith-number/O7KZVJB2V37MX5EHPZFN47MW2M/graph.json","events_json":"https://pith.science/api/pith-number/O7KZVJB2V37MX5EHPZFN47MW2M/events.json","paper":"https://pith.science/paper/O7KZVJB2"},"agent_actions":{"view_html":"https://pith.science/pith/O7KZVJB2V37MX5EHPZFN47MW2M","download_json":"https://pith.science/pith/O7KZVJB2V37MX5EHPZFN47MW2M.json","view_paper":"https://pith.science/paper/O7KZVJB2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.20339&json=true","fetch_graph":"https://pith.science/api/pith-number/O7KZVJB2V37MX5EHPZFN47MW2M/graph.json","fetch_events":"https://pith.science/api/pith-number/O7KZVJB2V37MX5EHPZFN47MW2M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O7KZVJB2V37MX5EHPZFN47MW2M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O7KZVJB2V37MX5EHPZFN47MW2M/action/storage_attestation","attest_author":"https://pith.science/pith/O7KZVJB2V37MX5EHPZFN47MW2M/action/author_attestation","sign_citation":"https://pith.science/pith/O7KZVJB2V37MX5EHPZFN47MW2M/action/citation_signature","submit_replication":"https://pith.science/pith/O7KZVJB2V37MX5EHPZFN47MW2M/action/replication_record"}},"created_at":"2026-07-05T08:25:24.553764+00:00","updated_at":"2026-07-05T08:25:24.553764+00:00"}