{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7BBVQLT4LUIUCBXJUHJSBM6WEJ","short_pith_number":"pith:7BBVQLT4","schema_version":"1.0","canonical_sha256":"f843582e7c5d114106e9a1d320b3d62260332911a4e6a7dc085df75bdcfc64f7","source":{"kind":"arxiv","id":"2501.09532","version":2},"attestation_state":"computed","paper":{"title":"AdaFV: Rethinking of Visual-Language alignment for VLM acceleration","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongwei Du, Jiayi Han, Liang Du, Weibo Zheng, Xiangguo Zhou, Yiwen Wu","submitted_at":"2025-01-16T13:34:33Z","abstract_excerpt":"The success of VLMs often relies on the dynamic high-resolution schema that adaptively augments the input images to multiple crops, so that the details of the images can be retained. However, such approaches result in a large number of redundant visual tokens, thus significantly reducing the efficiency of the VLMs. To improve the VLMs' efficiency without introducing extra training costs, many research works are proposed to reduce the visual tokens by filtering the uninformative visual tokens or aggregating their information. Some approaches propose to reduce the visual tokens according to the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.09532","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-16T13:34:33Z","cross_cats_sorted":[],"title_canon_sha256":"cec0840a797b468f07efecf74a8f88555b3ca767809fa3e927c442054f588d96","abstract_canon_sha256":"aad9ae601cb8c583635e110b66637aedae3e628f0632fa8ef862f2d3437c8780"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:27.915350Z","signature_b64":"JR25lCCAv1JgedpkP7tOj83+5nm/ybGrv8sd+MRtmQHaBzC7nO2uvJ73srjFeBO0si2NuZh2Gjbc4MX1VJesDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f843582e7c5d114106e9a1d320b3d62260332911a4e6a7dc085df75bdcfc64f7","last_reissued_at":"2026-07-05T10:08:27.914890Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:27.914890Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AdaFV: Rethinking of Visual-Language alignment for VLM acceleration","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongwei Du, Jiayi Han, Liang Du, Weibo Zheng, Xiangguo Zhou, Yiwen Wu","submitted_at":"2025-01-16T13:34:33Z","abstract_excerpt":"The success of VLMs often relies on the dynamic high-resolution schema that adaptively augments the input images to multiple crops, so that the details of the images can be retained. However, such approaches result in a large number of redundant visual tokens, thus significantly reducing the efficiency of the VLMs. To improve the VLMs' efficiency without introducing extra training costs, many research works are proposed to reduce the visual tokens by filtering the uninformative visual tokens or aggregating their information. Some approaches propose to reduce the visual tokens according to the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.09532","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.09532/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.09532","created_at":"2026-07-05T10:08:27.914942+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.09532v2","created_at":"2026-07-05T10:08:27.914942+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.09532","created_at":"2026-07-05T10:08:27.914942+00:00"},{"alias_kind":"pith_short_12","alias_value":"7BBVQLT4LUIU","created_at":"2026-07-05T10:08:27.914942+00:00"},{"alias_kind":"pith_short_16","alias_value":"7BBVQLT4LUIUCBXJ","created_at":"2026-07-05T10:08:27.914942+00:00"},{"alias_kind":"pith_short_8","alias_value":"7BBVQLT4","created_at":"2026-07-05T10:08:27.914942+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11864","citing_title":"Very Efficient Listwise Multimodal Reranking for Long Documents","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11122","citing_title":"Semantic-Geometric Dual Compression: Training-Free Visual Token Reduction for Ultra-High-Resolution Remote Sensing Understanding","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7BBVQLT4LUIUCBXJUHJSBM6WEJ","json":"https://pith.science/pith/7BBVQLT4LUIUCBXJUHJSBM6WEJ.json","graph_json":"https://pith.science/api/pith-number/7BBVQLT4LUIUCBXJUHJSBM6WEJ/graph.json","events_json":"https://pith.science/api/pith-number/7BBVQLT4LUIUCBXJUHJSBM6WEJ/events.json","paper":"https://pith.science/paper/7BBVQLT4"},"agent_actions":{"view_html":"https://pith.science/pith/7BBVQLT4LUIUCBXJUHJSBM6WEJ","download_json":"https://pith.science/pith/7BBVQLT4LUIUCBXJUHJSBM6WEJ.json","view_paper":"https://pith.science/paper/7BBVQLT4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.09532&json=true","fetch_graph":"https://pith.science/api/pith-number/7BBVQLT4LUIUCBXJUHJSBM6WEJ/graph.json","fetch_events":"https://pith.science/api/pith-number/7BBVQLT4LUIUCBXJUHJSBM6WEJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7BBVQLT4LUIUCBXJUHJSBM6WEJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7BBVQLT4LUIUCBXJUHJSBM6WEJ/action/storage_attestation","attest_author":"https://pith.science/pith/7BBVQLT4LUIUCBXJUHJSBM6WEJ/action/author_attestation","sign_citation":"https://pith.science/pith/7BBVQLT4LUIUCBXJUHJSBM6WEJ/action/citation_signature","submit_replication":"https://pith.science/pith/7BBVQLT4LUIUCBXJUHJSBM6WEJ/action/replication_record"}},"created_at":"2026-07-05T10:08:27.914942+00:00","updated_at":"2026-07-05T10:08:27.914942+00:00"}