{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KOICFV5QES2PDUDRBRESSYB45Z","short_pith_number":"pith:KOICFV5Q","schema_version":"1.0","canonical_sha256":"539022d7b024b4f1d0710c4929603cee7f1b1daa1b268471e79e60b1fc02733e","source":{"kind":"arxiv","id":"2507.08000","version":1},"attestation_state":"computed","paper":{"title":"Impact of Pretraining Word Co-occurrence on Compositional Generalization in Multimodal Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Helen Qu, Sang Michael Xie","submitted_at":"2025-07-10T17:59:59Z","abstract_excerpt":"CLIP and large multimodal models (LMMs) have better accuracy on examples involving concepts that are highly represented in the training data. However, the role of concept combinations in the training data on compositional generalization is largely unclear -- for instance, how does accuracy vary when a common object appears in an uncommon pairing with another object? In this paper, we investigate how word co-occurrence statistics in the pretraining dataset (a proxy for co-occurrence of visual concepts) impacts CLIP/LMM performance. To disentangle the effects of word co-occurrence frequencies fr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.08000","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-10T17:59:59Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ec73d38550100cd4a1c88ab5c7d736a213ba5e1f90ad2ac951adf8508960b9d0","abstract_canon_sha256":"76abce125a2cdac2827caee9c27ae8abc879ead310df75ab590f2371c17f0682"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:35:08.393711Z","signature_b64":"bUEQqzegPqk7JBBvw7IFqwN31DRIybUAvOftvSX+j2dRbqHVUffG+LoK54TlWIuTEPyjrPKeZqyRHPnRx9t3Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"539022d7b024b4f1d0710c4929603cee7f1b1daa1b268471e79e60b1fc02733e","last_reissued_at":"2026-07-05T11:35:08.393235Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:35:08.393235Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Impact of Pretraining Word Co-occurrence on Compositional Generalization in Multimodal Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Helen Qu, Sang Michael Xie","submitted_at":"2025-07-10T17:59:59Z","abstract_excerpt":"CLIP and large multimodal models (LMMs) have better accuracy on examples involving concepts that are highly represented in the training data. However, the role of concept combinations in the training data on compositional generalization is largely unclear -- for instance, how does accuracy vary when a common object appears in an uncommon pairing with another object? In this paper, we investigate how word co-occurrence statistics in the pretraining dataset (a proxy for co-occurrence of visual concepts) impacts CLIP/LMM performance. To disentangle the effects of word co-occurrence frequencies fr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.08000","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.08000/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.08000","created_at":"2026-07-05T11:35:08.393291+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.08000v1","created_at":"2026-07-05T11:35:08.393291+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.08000","created_at":"2026-07-05T11:35:08.393291+00:00"},{"alias_kind":"pith_short_12","alias_value":"KOICFV5QES2P","created_at":"2026-07-05T11:35:08.393291+00:00"},{"alias_kind":"pith_short_16","alias_value":"KOICFV5QES2PDUDR","created_at":"2026-07-05T11:35:08.393291+00:00"},{"alias_kind":"pith_short_8","alias_value":"KOICFV5Q","created_at":"2026-07-05T11:35:08.393291+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KOICFV5QES2PDUDRBRESSYB45Z","json":"https://pith.science/pith/KOICFV5QES2PDUDRBRESSYB45Z.json","graph_json":"https://pith.science/api/pith-number/KOICFV5QES2PDUDRBRESSYB45Z/graph.json","events_json":"https://pith.science/api/pith-number/KOICFV5QES2PDUDRBRESSYB45Z/events.json","paper":"https://pith.science/paper/KOICFV5Q"},"agent_actions":{"view_html":"https://pith.science/pith/KOICFV5QES2PDUDRBRESSYB45Z","download_json":"https://pith.science/pith/KOICFV5QES2PDUDRBRESSYB45Z.json","view_paper":"https://pith.science/paper/KOICFV5Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.08000&json=true","fetch_graph":"https://pith.science/api/pith-number/KOICFV5QES2PDUDRBRESSYB45Z/graph.json","fetch_events":"https://pith.science/api/pith-number/KOICFV5QES2PDUDRBRESSYB45Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KOICFV5QES2PDUDRBRESSYB45Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KOICFV5QES2PDUDRBRESSYB45Z/action/storage_attestation","attest_author":"https://pith.science/pith/KOICFV5QES2PDUDRBRESSYB45Z/action/author_attestation","sign_citation":"https://pith.science/pith/KOICFV5QES2PDUDRBRESSYB45Z/action/citation_signature","submit_replication":"https://pith.science/pith/KOICFV5QES2PDUDRBRESSYB45Z/action/replication_record"}},"created_at":"2026-07-05T11:35:08.393291+00:00","updated_at":"2026-07-05T11:35:08.393291+00:00"}