{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J2INVTDXAMDWAYP25GQO24AUWL","short_pith_number":"pith:J2INVTDX","schema_version":"1.0","canonical_sha256":"4e90dacc7703076061fae9a0ed7014b2cd966b95a92106c6a8280f9ae77ead0e","source":{"kind":"arxiv","id":"2411.14164","version":1},"attestation_state":"computed","paper":{"title":"FoPru: Focal Pruning for Efficient Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jing Li, Lechao Cheng, Lei Jiang, Tongxuan Liu, Weizhe Huang, Xiaohua Xu, Yuting Zeng","submitted_at":"2024-11-21T14:22:38Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) represent a significant advancement toward achieving superior multimodal capabilities by enabling powerful Large Language Models (LLMs) to understand visual input. Typically, LVLMs utilize visual encoders, such as CLIP, to transform images into visual tokens, which are then aligned with textual tokens through projection layers before being input into the LLM for inference. Although existing LVLMs have achieved significant success, their inference efficiency is still limited by the substantial number of visual tokens and the potential redundancy among them. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.14164","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-21T14:22:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1780360e93cd48afae2d37c4154444d368ff23db4232fc59e4ec9d2e71000e8f","abstract_canon_sha256":"a38408cb8f8ddd5f427468018753e4a9fb263538341f7c08582fad58790785af"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:42.626879Z","signature_b64":"uX0vzB6Qd2Sl7GH8YvR1hVXdckb8WwS19hlaqeC68u1F8BTtR4GFfpgvZgRIrQTrww4CbRaJDLCbBFforLxsCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e90dacc7703076061fae9a0ed7014b2cd966b95a92106c6a8280f9ae77ead0e","last_reissued_at":"2026-07-05T09:38:42.626393Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:42.626393Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FoPru: Focal Pruning for Efficient Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jing Li, Lechao Cheng, Lei Jiang, Tongxuan Liu, Weizhe Huang, Xiaohua Xu, Yuting Zeng","submitted_at":"2024-11-21T14:22:38Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) represent a significant advancement toward achieving superior multimodal capabilities by enabling powerful Large Language Models (LLMs) to understand visual input. Typically, LVLMs utilize visual encoders, such as CLIP, to transform images into visual tokens, which are then aligned with textual tokens through projection layers before being input into the LLM for inference. Although existing LVLMs have achieved significant success, their inference efficiency is still limited by the substantial number of visual tokens and the potential redundancy among them. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.14164","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.14164/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.14164","created_at":"2026-07-05T09:38:42.626455+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.14164v1","created_at":"2026-07-05T09:38:42.626455+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.14164","created_at":"2026-07-05T09:38:42.626455+00:00"},{"alias_kind":"pith_short_12","alias_value":"J2INVTDXAMDW","created_at":"2026-07-05T09:38:42.626455+00:00"},{"alias_kind":"pith_short_16","alias_value":"J2INVTDXAMDWAYP2","created_at":"2026-07-05T09:38:42.626455+00:00"},{"alias_kind":"pith_short_8","alias_value":"J2INVTDX","created_at":"2026-07-05T09:38:42.626455+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12412","citing_title":"Reroute, Don't Remove: Recoverable Visual Token Routing for Vision-Language Models","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J2INVTDXAMDWAYP25GQO24AUWL","json":"https://pith.science/pith/J2INVTDXAMDWAYP25GQO24AUWL.json","graph_json":"https://pith.science/api/pith-number/J2INVTDXAMDWAYP25GQO24AUWL/graph.json","events_json":"https://pith.science/api/pith-number/J2INVTDXAMDWAYP25GQO24AUWL/events.json","paper":"https://pith.science/paper/J2INVTDX"},"agent_actions":{"view_html":"https://pith.science/pith/J2INVTDXAMDWAYP25GQO24AUWL","download_json":"https://pith.science/pith/J2INVTDXAMDWAYP25GQO24AUWL.json","view_paper":"https://pith.science/paper/J2INVTDX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.14164&json=true","fetch_graph":"https://pith.science/api/pith-number/J2INVTDXAMDWAYP25GQO24AUWL/graph.json","fetch_events":"https://pith.science/api/pith-number/J2INVTDXAMDWAYP25GQO24AUWL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J2INVTDXAMDWAYP25GQO24AUWL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J2INVTDXAMDWAYP25GQO24AUWL/action/storage_attestation","attest_author":"https://pith.science/pith/J2INVTDXAMDWAYP25GQO24AUWL/action/author_attestation","sign_citation":"https://pith.science/pith/J2INVTDXAMDWAYP25GQO24AUWL/action/citation_signature","submit_replication":"https://pith.science/pith/J2INVTDXAMDWAYP25GQO24AUWL/action/replication_record"}},"created_at":"2026-07-05T09:38:42.626455+00:00","updated_at":"2026-07-05T09:38:42.626455+00:00"}