{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ADEZ4S3VE6XRR35ARAEJXKZLYM","short_pith_number":"pith:ADEZ4S3V","schema_version":"1.0","canonical_sha256":"00c99e4b7527af18efa088089bab2bc31dce916d2a78b32c1efd5e04490a5b2e","source":{"kind":"arxiv","id":"2204.10496","version":2},"attestation_state":"computed","paper":{"title":"Multimodal Adaptive Distillation for Leveraging Unimodal Encoders for Vision-Language Tasks","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Bin Xiao, Haoxuan You, Jianwei Yang, Kai-Wei Chang, Luowei Zhou, Lu Yuan, Noel Codella, Shih-Fu Chang, Xiyang Dai, Yen-Chun Chen, Zhecan Wang","submitted_at":"2022-04-22T04:41:04Z","abstract_excerpt":"Cross-modal encoders for vision-language (VL) tasks are often pretrained with carefully curated vision-language datasets. While these datasets reach an order of 10 million samples, the labor cost is prohibitive to scale further. Conversely, unimodal encoders are pretrained with simpler annotations that are less cost-prohibitive, achieving scales of hundreds of millions to billions. As a result, unimodal encoders have achieved state-of-art (SOTA) on many downstream tasks. However, challenges remain when applying to VL tasks. The pretraining data is not optimal for cross-modal architectures and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.10496","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2022-04-22T04:41:04Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG","cs.MM"],"title_canon_sha256":"16649aa3117c350d9788e0444431250b262d03a3922bc2848b9fa1360bf5b953","abstract_canon_sha256":"91e9686094ab8a55b5133095ff0caac16c9c64b80f36393e757560941abe0e13"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:18:42.066824Z","signature_b64":"Rfo+4vrSH8dCN1m6hxj7xkvJwuNTjgcWMPlpihwkd4qpJ6ld9c4b/zCrALr1aXU6Yg9M9apJ5jUoF6laA0r+Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"00c99e4b7527af18efa088089bab2bc31dce916d2a78b32c1efd5e04490a5b2e","last_reissued_at":"2026-07-05T04:18:42.066308Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:18:42.066308Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Adaptive Distillation for Leveraging Unimodal Encoders for Vision-Language Tasks","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Bin Xiao, Haoxuan You, Jianwei Yang, Kai-Wei Chang, Luowei Zhou, Lu Yuan, Noel Codella, Shih-Fu Chang, Xiyang Dai, Yen-Chun Chen, Zhecan Wang","submitted_at":"2022-04-22T04:41:04Z","abstract_excerpt":"Cross-modal encoders for vision-language (VL) tasks are often pretrained with carefully curated vision-language datasets. While these datasets reach an order of 10 million samples, the labor cost is prohibitive to scale further. Conversely, unimodal encoders are pretrained with simpler annotations that are less cost-prohibitive, achieving scales of hundreds of millions to billions. As a result, unimodal encoders have achieved state-of-art (SOTA) on many downstream tasks. However, challenges remain when applying to VL tasks. The pretraining data is not optimal for cross-modal architectures and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.10496","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.10496/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.10496","created_at":"2026-07-05T04:18:42.066379+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.10496v2","created_at":"2026-07-05T04:18:42.066379+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.10496","created_at":"2026-07-05T04:18:42.066379+00:00"},{"alias_kind":"pith_short_12","alias_value":"ADEZ4S3VE6XR","created_at":"2026-07-05T04:18:42.066379+00:00"},{"alias_kind":"pith_short_16","alias_value":"ADEZ4S3VE6XRR35A","created_at":"2026-07-05T04:18:42.066379+00:00"},{"alias_kind":"pith_short_8","alias_value":"ADEZ4S3V","created_at":"2026-07-05T04:18:42.066379+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12282","citing_title":"Large-Small Model Collaboration for Farmland Semantic Change Detection","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":205,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ADEZ4S3VE6XRR35ARAEJXKZLYM","json":"https://pith.science/pith/ADEZ4S3VE6XRR35ARAEJXKZLYM.json","graph_json":"https://pith.science/api/pith-number/ADEZ4S3VE6XRR35ARAEJXKZLYM/graph.json","events_json":"https://pith.science/api/pith-number/ADEZ4S3VE6XRR35ARAEJXKZLYM/events.json","paper":"https://pith.science/paper/ADEZ4S3V"},"agent_actions":{"view_html":"https://pith.science/pith/ADEZ4S3VE6XRR35ARAEJXKZLYM","download_json":"https://pith.science/pith/ADEZ4S3VE6XRR35ARAEJXKZLYM.json","view_paper":"https://pith.science/paper/ADEZ4S3V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.10496&json=true","fetch_graph":"https://pith.science/api/pith-number/ADEZ4S3VE6XRR35ARAEJXKZLYM/graph.json","fetch_events":"https://pith.science/api/pith-number/ADEZ4S3VE6XRR35ARAEJXKZLYM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ADEZ4S3VE6XRR35ARAEJXKZLYM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ADEZ4S3VE6XRR35ARAEJXKZLYM/action/storage_attestation","attest_author":"https://pith.science/pith/ADEZ4S3VE6XRR35ARAEJXKZLYM/action/author_attestation","sign_citation":"https://pith.science/pith/ADEZ4S3VE6XRR35ARAEJXKZLYM/action/citation_signature","submit_replication":"https://pith.science/pith/ADEZ4S3VE6XRR35ARAEJXKZLYM/action/replication_record"}},"created_at":"2026-07-05T04:18:42.066379+00:00","updated_at":"2026-07-05T04:18:42.066379+00:00"}