{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UYCD3DJD7QSWF44U5PVB4UQPGC","short_pith_number":"pith:UYCD3DJD","schema_version":"1.0","canonical_sha256":"a6043d8d23fc2562f394ebea1e520f30a826b182de809dc44bfcf9463e4f1e19","source":{"kind":"arxiv","id":"2410.11829","version":1},"attestation_state":"computed","paper":{"title":"MMFuser: Multimodal Multi-Layer Feature Fuser for Fine-Grained Vision-Language Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Danhuai Zhao, Guangchen Shi, Tong Lu, Wenhai Wang, Yangzhou Liu, Yue Cao, Zhe Chen","submitted_at":"2024-10-15T17:55:22Z","abstract_excerpt":"Despite significant advancements in Multimodal Large Language Models (MLLMs) for understanding complex human intentions through cross-modal interactions, capturing intricate image details remains challenging. Previous methods integrating multiple vision encoders to enhance visual detail introduce redundancy and computational overhead. We observe that most MLLMs utilize only the last-layer feature map of the vision encoder for visual representation, neglecting the rich fine-grained information in shallow feature maps. To address this issue, we propose \\modelname, a simple yet effective multi-la"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.11829","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-15T17:55:22Z","cross_cats_sorted":[],"title_canon_sha256":"ef634e94ba273237d83b22c6dd923fca7242637a482f2c7836d5086ab2b0efb3","abstract_canon_sha256":"2730f2c26926754b4a617df0a360fac10fc9a5a59a07a2a803f1dc4794c4556c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:59.553872Z","signature_b64":"t5KUxuVA17616PK9scdBrziASm5R31me7BGkJf+BXFpVCKgRM6GUyYz3kDW2M5T50YxhD9FmUlUN3h1cIMVXAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6043d8d23fc2562f394ebea1e520f30a826b182de809dc44bfcf9463e4f1e19","last_reissued_at":"2026-07-05T09:20:59.553232Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:59.553232Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMFuser: Multimodal Multi-Layer Feature Fuser for Fine-Grained Vision-Language Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Danhuai Zhao, Guangchen Shi, Tong Lu, Wenhai Wang, Yangzhou Liu, Yue Cao, Zhe Chen","submitted_at":"2024-10-15T17:55:22Z","abstract_excerpt":"Despite significant advancements in Multimodal Large Language Models (MLLMs) for understanding complex human intentions through cross-modal interactions, capturing intricate image details remains challenging. Previous methods integrating multiple vision encoders to enhance visual detail introduce redundancy and computational overhead. We observe that most MLLMs utilize only the last-layer feature map of the vision encoder for visual representation, neglecting the rich fine-grained information in shallow feature maps. To address this issue, we propose \\modelname, a simple yet effective multi-la"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.11829","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.11829/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.11829","created_at":"2026-07-05T09:20:59.553300+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.11829v1","created_at":"2026-07-05T09:20:59.553300+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.11829","created_at":"2026-07-05T09:20:59.553300+00:00"},{"alias_kind":"pith_short_12","alias_value":"UYCD3DJD7QSW","created_at":"2026-07-05T09:20:59.553300+00:00"},{"alias_kind":"pith_short_16","alias_value":"UYCD3DJD7QSWF44U","created_at":"2026-07-05T09:20:59.553300+00:00"},{"alias_kind":"pith_short_8","alias_value":"UYCD3DJD","created_at":"2026-07-05T09:20:59.553300+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.00655","citing_title":"Mema: Memory-Augmented Adapter for Enhanced Vision-Language Understanding","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10780","citing_title":"Beyond the Last Layer: Multi-Layer Representation Fusion for Visual Tokenization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10780","citing_title":"Beyond the Last Layer: Multi-Layer Representation Fusion for Visual Tokenization","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UYCD3DJD7QSWF44U5PVB4UQPGC","json":"https://pith.science/pith/UYCD3DJD7QSWF44U5PVB4UQPGC.json","graph_json":"https://pith.science/api/pith-number/UYCD3DJD7QSWF44U5PVB4UQPGC/graph.json","events_json":"https://pith.science/api/pith-number/UYCD3DJD7QSWF44U5PVB4UQPGC/events.json","paper":"https://pith.science/paper/UYCD3DJD"},"agent_actions":{"view_html":"https://pith.science/pith/UYCD3DJD7QSWF44U5PVB4UQPGC","download_json":"https://pith.science/pith/UYCD3DJD7QSWF44U5PVB4UQPGC.json","view_paper":"https://pith.science/paper/UYCD3DJD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.11829&json=true","fetch_graph":"https://pith.science/api/pith-number/UYCD3DJD7QSWF44U5PVB4UQPGC/graph.json","fetch_events":"https://pith.science/api/pith-number/UYCD3DJD7QSWF44U5PVB4UQPGC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UYCD3DJD7QSWF44U5PVB4UQPGC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UYCD3DJD7QSWF44U5PVB4UQPGC/action/storage_attestation","attest_author":"https://pith.science/pith/UYCD3DJD7QSWF44U5PVB4UQPGC/action/author_attestation","sign_citation":"https://pith.science/pith/UYCD3DJD7QSWF44U5PVB4UQPGC/action/citation_signature","submit_replication":"https://pith.science/pith/UYCD3DJD7QSWF44U5PVB4UQPGC/action/replication_record"}},"created_at":"2026-07-05T09:20:59.553300+00:00","updated_at":"2026-07-05T09:20:59.553300+00:00"}