{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:EPOMEZ4EBE35CTYVDCU6HMMKPS","short_pith_number":"pith:EPOMEZ4E","schema_version":"1.0","canonical_sha256":"23dcc267840937d14f1518a9e3b18a7c857323f69f7227c33f39b1ed645968ae","source":{"kind":"arxiv","id":"2308.12898","version":2},"attestation_state":"computed","paper":{"title":"Can Linguistic Knowledge Improve Multimodal Alignment in Vision-Language Pretraining?","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.MM","authors_text":"Changxing Ding, Fei Wang, Jun Rao, Liang Ding, Li Shen, Ye Liu","submitted_at":"2023-08-24T16:17:40Z","abstract_excerpt":"The multimedia community has shown a significant interest in perceiving and representing the physical world with multimodal pretrained neural network models, and among them, the visual-language pertaining (VLP) is, currently, the most captivating topic. However, there have been few endeavors dedicated to the exploration of 1) whether essential linguistic knowledge (e.g., semantics and syntax) can be extracted during VLP, and 2) how such linguistic knowledge impact or enhance the multimodal alignment. In response, here we aim to elucidate the impact of comprehensive linguistic knowledge, includ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.12898","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.MM","submitted_at":"2023-08-24T16:17:40Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"c201beacdb512df82753d0b6832ecda1202070651610b2459a097a878f67520a","abstract_canon_sha256":"e145052b4b4d5e073e1e338e1efa6a9ae674c2baa1fe627df6cb83c95ec4357f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:44:36.647948Z","signature_b64":"STtj70HFMi4Sl6w7ToTEVygnzrFSZrX/Y4Z+XqRsTROTW59YhxNY0UQRzSqlAjZ9M2+/oPsSv2r7hp64+JlwAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23dcc267840937d14f1518a9e3b18a7c857323f69f7227c33f39b1ed645968ae","last_reissued_at":"2026-07-05T06:44:36.647524Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:44:36.647524Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Linguistic Knowledge Improve Multimodal Alignment in Vision-Language Pretraining?","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.MM","authors_text":"Changxing Ding, Fei Wang, Jun Rao, Liang Ding, Li Shen, Ye Liu","submitted_at":"2023-08-24T16:17:40Z","abstract_excerpt":"The multimedia community has shown a significant interest in perceiving and representing the physical world with multimodal pretrained neural network models, and among them, the visual-language pertaining (VLP) is, currently, the most captivating topic. However, there have been few endeavors dedicated to the exploration of 1) whether essential linguistic knowledge (e.g., semantics and syntax) can be extracted during VLP, and 2) how such linguistic knowledge impact or enhance the multimodal alignment. In response, here we aim to elucidate the impact of comprehensive linguistic knowledge, includ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.12898","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.12898/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.12898","created_at":"2026-07-05T06:44:36.647586+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.12898v2","created_at":"2026-07-05T06:44:36.647586+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.12898","created_at":"2026-07-05T06:44:36.647586+00:00"},{"alias_kind":"pith_short_12","alias_value":"EPOMEZ4EBE35","created_at":"2026-07-05T06:44:36.647586+00:00"},{"alias_kind":"pith_short_16","alias_value":"EPOMEZ4EBE35CTYV","created_at":"2026-07-05T06:44:36.647586+00:00"},{"alias_kind":"pith_short_8","alias_value":"EPOMEZ4E","created_at":"2026-07-05T06:44:36.647586+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.08111","citing_title":"Seeing Syntax: Uncovering Syntactic Learning Limitations in Vision-Language Models","ref_index":43,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EPOMEZ4EBE35CTYVDCU6HMMKPS","json":"https://pith.science/pith/EPOMEZ4EBE35CTYVDCU6HMMKPS.json","graph_json":"https://pith.science/api/pith-number/EPOMEZ4EBE35CTYVDCU6HMMKPS/graph.json","events_json":"https://pith.science/api/pith-number/EPOMEZ4EBE35CTYVDCU6HMMKPS/events.json","paper":"https://pith.science/paper/EPOMEZ4E"},"agent_actions":{"view_html":"https://pith.science/pith/EPOMEZ4EBE35CTYVDCU6HMMKPS","download_json":"https://pith.science/pith/EPOMEZ4EBE35CTYVDCU6HMMKPS.json","view_paper":"https://pith.science/paper/EPOMEZ4E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.12898&json=true","fetch_graph":"https://pith.science/api/pith-number/EPOMEZ4EBE35CTYVDCU6HMMKPS/graph.json","fetch_events":"https://pith.science/api/pith-number/EPOMEZ4EBE35CTYVDCU6HMMKPS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EPOMEZ4EBE35CTYVDCU6HMMKPS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EPOMEZ4EBE35CTYVDCU6HMMKPS/action/storage_attestation","attest_author":"https://pith.science/pith/EPOMEZ4EBE35CTYVDCU6HMMKPS/action/author_attestation","sign_citation":"https://pith.science/pith/EPOMEZ4EBE35CTYVDCU6HMMKPS/action/citation_signature","submit_replication":"https://pith.science/pith/EPOMEZ4EBE35CTYVDCU6HMMKPS/action/replication_record"}},"created_at":"2026-07-05T06:44:36.647586+00:00","updated_at":"2026-07-05T06:44:36.647586+00:00"}