{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BWSDD7SIEWJMTLGL3EEUDMU4XQ","short_pith_number":"pith:BWSDD7SI","schema_version":"1.0","canonical_sha256":"0da431fe482592c9accbd90941b29cbc0fa5a03c0b2a95a4a8a13e767f61d845","source":{"kind":"arxiv","id":"2411.11927","version":3},"attestation_state":"computed","paper":{"title":"FLAME: Frozen Large Language Models Enable Data-Efficient Language-Image Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anjia Cao, Xing Wei, Zhiheng Ma","submitted_at":"2024-11-18T09:19:30Z","abstract_excerpt":"Language-image pre-training faces significant challenges due to limited data in specific formats and the constrained capacities of text encoders. While prevailing methods attempt to address these issues through data augmentation and architecture modifications, they continue to struggle with processing long-form text inputs, and the inherent limitations of traditional CLIP text encoders lead to suboptimal downstream generalization. In this paper, we propose FLAME (Frozen Large lAnguage Models Enable data-efficient language-image pre-training) that leverages frozen large language models as text "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.11927","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-18T09:19:30Z","cross_cats_sorted":[],"title_canon_sha256":"b011ff4ecb8f7ca89858b93d58094be09b9fdb98ba85f20c3ebf6a05fcd69ab0","abstract_canon_sha256":"bec3db9136a17019d37b667c41fb7efc67015dbcd6c7d48bde2e223f4f92f4d5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:39.567675Z","signature_b64":"zLij1gTlRCXR7mESMhBXve0RI2C5IRL+kJpiHYpgrP7ojVFEjUZkVrXrxa0lXUyU+EdxoFHHTnNZIij9OpdYCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0da431fe482592c9accbd90941b29cbc0fa5a03c0b2a95a4a8a13e767f61d845","last_reissued_at":"2026-07-05T10:54:39.567219Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:39.567219Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FLAME: Frozen Large Language Models Enable Data-Efficient Language-Image Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anjia Cao, Xing Wei, Zhiheng Ma","submitted_at":"2024-11-18T09:19:30Z","abstract_excerpt":"Language-image pre-training faces significant challenges due to limited data in specific formats and the constrained capacities of text encoders. While prevailing methods attempt to address these issues through data augmentation and architecture modifications, they continue to struggle with processing long-form text inputs, and the inherent limitations of traditional CLIP text encoders lead to suboptimal downstream generalization. In this paper, we propose FLAME (Frozen Large lAnguage Models Enable data-efficient language-image pre-training) that leverages frozen large language models as text "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.11927","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.11927/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.11927","created_at":"2026-07-05T10:54:39.567276+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.11927v3","created_at":"2026-07-05T10:54:39.567276+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.11927","created_at":"2026-07-05T10:54:39.567276+00:00"},{"alias_kind":"pith_short_12","alias_value":"BWSDD7SIEWJM","created_at":"2026-07-05T10:54:39.567276+00:00"},{"alias_kind":"pith_short_16","alias_value":"BWSDD7SIEWJMTLGL","created_at":"2026-07-05T10:54:39.567276+00:00"},{"alias_kind":"pith_short_8","alias_value":"BWSDD7SI","created_at":"2026-07-05T10:54:39.567276+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BWSDD7SIEWJMTLGL3EEUDMU4XQ","json":"https://pith.science/pith/BWSDD7SIEWJMTLGL3EEUDMU4XQ.json","graph_json":"https://pith.science/api/pith-number/BWSDD7SIEWJMTLGL3EEUDMU4XQ/graph.json","events_json":"https://pith.science/api/pith-number/BWSDD7SIEWJMTLGL3EEUDMU4XQ/events.json","paper":"https://pith.science/paper/BWSDD7SI"},"agent_actions":{"view_html":"https://pith.science/pith/BWSDD7SIEWJMTLGL3EEUDMU4XQ","download_json":"https://pith.science/pith/BWSDD7SIEWJMTLGL3EEUDMU4XQ.json","view_paper":"https://pith.science/paper/BWSDD7SI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.11927&json=true","fetch_graph":"https://pith.science/api/pith-number/BWSDD7SIEWJMTLGL3EEUDMU4XQ/graph.json","fetch_events":"https://pith.science/api/pith-number/BWSDD7SIEWJMTLGL3EEUDMU4XQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BWSDD7SIEWJMTLGL3EEUDMU4XQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BWSDD7SIEWJMTLGL3EEUDMU4XQ/action/storage_attestation","attest_author":"https://pith.science/pith/BWSDD7SIEWJMTLGL3EEUDMU4XQ/action/author_attestation","sign_citation":"https://pith.science/pith/BWSDD7SIEWJMTLGL3EEUDMU4XQ/action/citation_signature","submit_replication":"https://pith.science/pith/BWSDD7SIEWJMTLGL3EEUDMU4XQ/action/replication_record"}},"created_at":"2026-07-05T10:54:39.567276+00:00","updated_at":"2026-07-05T10:54:39.567276+00:00"}