{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RCQIGMALHTDTVVXSIF3VYFT6DL","short_pith_number":"pith:RCQIGMAL","schema_version":"1.0","canonical_sha256":"88a083300b3cc73ad6f241775c167e1ad68977c94ec42d230f9b70bb957b861d","source":{"kind":"arxiv","id":"2408.14471","version":2},"attestation_state":"computed","paper":{"title":"A Practitioner's Guide to Continual Multimodal Pretraining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Ameya Prabhu, Karsten Roth, Matthias Bethge, Mehdi Cherti, Olivier H\\'enaff, Oriol Vinyals, Samuel Albanie, Sebastian Dziadzio, Vishaal Udandarao, Zeynep Akata","submitted_at":"2024-08-26T17:59:01Z","abstract_excerpt":"Multimodal foundation models serve numerous applications at the intersection of vision and language. Still, despite being pretrained on extensive data, they become outdated over time. To keep models updated, research into continual pretraining mainly explores scenarios with either (1) infrequent, indiscriminate updates on large-scale new data, or (2) frequent, sample-level updates. However, practical model deployment often operates in the gap between these two limit cases, as real-world applications often demand adaptation to specific subdomains, tasks or concepts -- spread over the entire, va"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.14471","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-26T17:59:01Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"9e564680bc3ae6f6fbf133b85db83a65308037e69a9e2032ef886459027c2787","abstract_canon_sha256":"d9d79654bab60da3e6e205fff730b50ffd6145f1b3c4dddca99c9effdafc681d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:13.910440Z","signature_b64":"+WYQ2EB8GqXr200BZqEFZJaeDxLugL5plJjv8eBUztzbtoCz+WNYNjlW4FaBWsqzAwIyn3Cj+g6VJMbCHThKDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"88a083300b3cc73ad6f241775c167e1ad68977c94ec42d230f9b70bb957b861d","last_reissued_at":"2026-07-05T09:45:13.910002Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:13.910002Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Practitioner's Guide to Continual Multimodal Pretraining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Ameya Prabhu, Karsten Roth, Matthias Bethge, Mehdi Cherti, Olivier H\\'enaff, Oriol Vinyals, Samuel Albanie, Sebastian Dziadzio, Vishaal Udandarao, Zeynep Akata","submitted_at":"2024-08-26T17:59:01Z","abstract_excerpt":"Multimodal foundation models serve numerous applications at the intersection of vision and language. Still, despite being pretrained on extensive data, they become outdated over time. To keep models updated, research into continual pretraining mainly explores scenarios with either (1) infrequent, indiscriminate updates on large-scale new data, or (2) frequent, sample-level updates. However, practical model deployment often operates in the gap between these two limit cases, as real-world applications often demand adaptation to specific subdomains, tasks or concepts -- spread over the entire, va"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.14471","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.14471/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.14471","created_at":"2026-07-05T09:45:13.910054+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.14471v2","created_at":"2026-07-05T09:45:13.910054+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.14471","created_at":"2026-07-05T09:45:13.910054+00:00"},{"alias_kind":"pith_short_12","alias_value":"RCQIGMALHTDT","created_at":"2026-07-05T09:45:13.910054+00:00"},{"alias_kind":"pith_short_16","alias_value":"RCQIGMALHTDTVVXS","created_at":"2026-07-05T09:45:13.910054+00:00"},{"alias_kind":"pith_short_8","alias_value":"RCQIGMAL","created_at":"2026-07-05T09:45:13.910054+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22999","citing_title":"Black-Box Continual Learning for Vision-Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":248,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02502","citing_title":"CRAM: Centroid-Routing and Adaptive MoE for Multimodal Continual Instruction Tuning","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02576","citing_title":"ProtoAda: Prototype-Guided Adaptive Adapter Expansion and Geometric Consolidation for Multimodal Continual Instruction Tuning","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":248,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24863","citing_title":"Rethinking Continual Learning for Speech and Audio: A Representation-Centric Taxonomy and Open Problems","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04227","citing_title":"Continual Learning for VLMs: A Survey and Taxonomy Beyond Forgetting","ref_index":68,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RCQIGMALHTDTVVXSIF3VYFT6DL","json":"https://pith.science/pith/RCQIGMALHTDTVVXSIF3VYFT6DL.json","graph_json":"https://pith.science/api/pith-number/RCQIGMALHTDTVVXSIF3VYFT6DL/graph.json","events_json":"https://pith.science/api/pith-number/RCQIGMALHTDTVVXSIF3VYFT6DL/events.json","paper":"https://pith.science/paper/RCQIGMAL"},"agent_actions":{"view_html":"https://pith.science/pith/RCQIGMALHTDTVVXSIF3VYFT6DL","download_json":"https://pith.science/pith/RCQIGMALHTDTVVXSIF3VYFT6DL.json","view_paper":"https://pith.science/paper/RCQIGMAL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.14471&json=true","fetch_graph":"https://pith.science/api/pith-number/RCQIGMALHTDTVVXSIF3VYFT6DL/graph.json","fetch_events":"https://pith.science/api/pith-number/RCQIGMALHTDTVVXSIF3VYFT6DL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RCQIGMALHTDTVVXSIF3VYFT6DL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RCQIGMALHTDTVVXSIF3VYFT6DL/action/storage_attestation","attest_author":"https://pith.science/pith/RCQIGMALHTDTVVXSIF3VYFT6DL/action/author_attestation","sign_citation":"https://pith.science/pith/RCQIGMALHTDTVVXSIF3VYFT6DL/action/citation_signature","submit_replication":"https://pith.science/pith/RCQIGMALHTDTVVXSIF3VYFT6DL/action/replication_record"}},"created_at":"2026-07-05T09:45:13.910054+00:00","updated_at":"2026-07-05T09:45:13.910054+00:00"}