{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MBL2AXXEF32PQCHRWDZYMOS34T","short_pith_number":"pith:MBL2AXXE","schema_version":"1.0","canonical_sha256":"6057a05ee42ef4f808f1b0f3863a5be4d56d9f6e68f1e198f3ffb36603d2bc5a","source":{"kind":"arxiv","id":"2409.14607","version":2},"attestation_state":"computed","paper":{"title":"Patch Ranking: Efficient CLIP by Learning to Rank Local Patches","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Cheng-En Wu, Jinhong Lin, Pedro Morgado, Yu Hen Hu","submitted_at":"2024-09-22T22:04:26Z","abstract_excerpt":"Contrastive image-text pre-trained models such as CLIP have shown remarkable adaptability to downstream tasks. However, they face challenges due to the high computational requirements of the Vision Transformer (ViT) backbone. Current strategies to boost ViT efficiency focus on pruning patch tokens but fall short in addressing the multimodal nature of CLIP and identifying the optimal subset of tokens for maximum performance. To address this, we propose greedy search methods to establish a \"Golden Ranking\" and introduce a lightweight predictor specifically trained to approximate this Ranking. To"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.14607","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-22T22:04:26Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"64c9ab7e16e237c0fc041454b167db4579158b9589d286ca00c3a808609bb5f0","abstract_canon_sha256":"b993de30e6f9cf568af46eeb64b02ce7176d37c8e735d71afdbb731c2cd89313"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:42.187806Z","signature_b64":"cpCj+RpUHHiGrZ/iqK4cbLzU3gabwDT3xU23NMG6Fsz14gimAnP/5ZNaFkZZS8WKK+YnzGGYCUJREkivsn3/BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6057a05ee42ef4f808f1b0f3863a5be4d56d9f6e68f1e198f3ffb36603d2bc5a","last_reissued_at":"2026-07-05T09:41:42.187320Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:42.187320Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Patch Ranking: Efficient CLIP by Learning to Rank Local Patches","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Cheng-En Wu, Jinhong Lin, Pedro Morgado, Yu Hen Hu","submitted_at":"2024-09-22T22:04:26Z","abstract_excerpt":"Contrastive image-text pre-trained models such as CLIP have shown remarkable adaptability to downstream tasks. However, they face challenges due to the high computational requirements of the Vision Transformer (ViT) backbone. Current strategies to boost ViT efficiency focus on pruning patch tokens but fall short in addressing the multimodal nature of CLIP and identifying the optimal subset of tokens for maximum performance. To address this, we propose greedy search methods to establish a \"Golden Ranking\" and introduce a lightweight predictor specifically trained to approximate this Ranking. To"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14607","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.14607/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.14607","created_at":"2026-07-05T09:41:42.187379+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.14607v2","created_at":"2026-07-05T09:41:42.187379+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14607","created_at":"2026-07-05T09:41:42.187379+00:00"},{"alias_kind":"pith_short_12","alias_value":"MBL2AXXEF32P","created_at":"2026-07-05T09:41:42.187379+00:00"},{"alias_kind":"pith_short_16","alias_value":"MBL2AXXEF32PQCHR","created_at":"2026-07-05T09:41:42.187379+00:00"},{"alias_kind":"pith_short_8","alias_value":"MBL2AXXE","created_at":"2026-07-05T09:41:42.187379+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MBL2AXXEF32PQCHRWDZYMOS34T","json":"https://pith.science/pith/MBL2AXXEF32PQCHRWDZYMOS34T.json","graph_json":"https://pith.science/api/pith-number/MBL2AXXEF32PQCHRWDZYMOS34T/graph.json","events_json":"https://pith.science/api/pith-number/MBL2AXXEF32PQCHRWDZYMOS34T/events.json","paper":"https://pith.science/paper/MBL2AXXE"},"agent_actions":{"view_html":"https://pith.science/pith/MBL2AXXEF32PQCHRWDZYMOS34T","download_json":"https://pith.science/pith/MBL2AXXEF32PQCHRWDZYMOS34T.json","view_paper":"https://pith.science/paper/MBL2AXXE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.14607&json=true","fetch_graph":"https://pith.science/api/pith-number/MBL2AXXEF32PQCHRWDZYMOS34T/graph.json","fetch_events":"https://pith.science/api/pith-number/MBL2AXXEF32PQCHRWDZYMOS34T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MBL2AXXEF32PQCHRWDZYMOS34T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MBL2AXXEF32PQCHRWDZYMOS34T/action/storage_attestation","attest_author":"https://pith.science/pith/MBL2AXXEF32PQCHRWDZYMOS34T/action/author_attestation","sign_citation":"https://pith.science/pith/MBL2AXXEF32PQCHRWDZYMOS34T/action/citation_signature","submit_replication":"https://pith.science/pith/MBL2AXXEF32PQCHRWDZYMOS34T/action/replication_record"}},"created_at":"2026-07-05T09:41:42.187379+00:00","updated_at":"2026-07-05T09:41:42.187379+00:00"}