{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XA4HPQGD5W6ZIRDH6YVMZYLBKG","short_pith_number":"pith:XA4HPQGD","schema_version":"1.0","canonical_sha256":"b83877c0c3edbd944467f62acce16151ad914d42945ccea4f30176e727a66ab7","source":{"kind":"arxiv","id":"2111.08276","version":3},"attestation_state":"computed","paper":{"title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Hang Li, Xinsong Zhang, Yan Zeng","submitted_at":"2021-11-16T07:55:26Z","abstract_excerpt":"Most existing methods in vision language pre-training rely on object-centric features extracted through object detection and make fine-grained alignments between the extracted features and texts. It is challenging for these methods to learn relations among multiple objects. To this end, we propose a new method called X-VLM to perform `multi-grained vision language pre-training.' The key to learning multi-grained alignments is to locate visual concepts in the image given the associated texts, and in the meantime align the texts with the visual concepts, where the alignments are in multi-granula"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.08276","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-11-16T07:55:26Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"3be38212acff811d80a66220850992e2f8f0da9a37b3f1c117066e5b57d9bbf8","abstract_canon_sha256":"390e3155d41d4fec30fb79da49b75c4b388dfc89039e735978c4ff50969e0849"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:28:23.241108Z","signature_b64":"oautuBjY4Q88qANbJrVa24HEDWzWAE4786+7mRFhqe6uuhnTMEMhgOFKVGtYuAYPZfxoR8p+kgqLYPiNL8hADQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b83877c0c3edbd944467f62acce16151ad914d42945ccea4f30176e727a66ab7","last_reissued_at":"2026-07-05T04:28:23.240580Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:28:23.240580Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Hang Li, Xinsong Zhang, Yan Zeng","submitted_at":"2021-11-16T07:55:26Z","abstract_excerpt":"Most existing methods in vision language pre-training rely on object-centric features extracted through object detection and make fine-grained alignments between the extracted features and texts. It is challenging for these methods to learn relations among multiple objects. To this end, we propose a new method called X-VLM to perform `multi-grained vision language pre-training.' The key to learning multi-grained alignments is to locate visual concepts in the image given the associated texts, and in the meantime align the texts with the visual concepts, where the alignments are in multi-granula"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.08276","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.08276/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.08276","created_at":"2026-07-05T04:28:23.240652+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.08276v3","created_at":"2026-07-05T04:28:23.240652+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.08276","created_at":"2026-07-05T04:28:23.240652+00:00"},{"alias_kind":"pith_short_12","alias_value":"XA4HPQGD5W6Z","created_at":"2026-07-05T04:28:23.240652+00:00"},{"alias_kind":"pith_short_16","alias_value":"XA4HPQGD5W6ZIRDH","created_at":"2026-07-05T04:28:23.240652+00:00"},{"alias_kind":"pith_short_8","alias_value":"XA4HPQGD","created_at":"2026-07-05T04:28:23.240652+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17030","citing_title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","ref_index":153,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14925","citing_title":"Road Maps as Free Geometric Priors: Weather-Invariant Drone Geo-Localization with GeoFuse","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00673","citing_title":"T-CLIP: Enabling Thermal Perception for Contrastive Language-Image Pretraining","ref_index":163,"is_internal_anchor":false},{"citing_arxiv_id":"2412.08110","citing_title":"The ART of Composition: Attention-Regularized Training for Compositional Visual Grounding","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2503.13821","citing_title":"Stitch-a-Demo: Video Demonstrations from Multistep Descriptions","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2310.00754","citing_title":"Analyzing and Mitigating Object Hallucination in Large Vision-Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2303.08128","citing_title":"ViperGPT: Visual Inference via Python Execution for Reasoning","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09921","citing_title":"WikiCLIP: An Efficient Contrastive Baseline for Open-domain Visual Entity Recognition","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08337","citing_title":"InstAP: Instance-Aware Vision-Language Pre-Train for Spatial-Temporal Understanding","ref_index":67,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XA4HPQGD5W6ZIRDH6YVMZYLBKG","json":"https://pith.science/pith/XA4HPQGD5W6ZIRDH6YVMZYLBKG.json","graph_json":"https://pith.science/api/pith-number/XA4HPQGD5W6ZIRDH6YVMZYLBKG/graph.json","events_json":"https://pith.science/api/pith-number/XA4HPQGD5W6ZIRDH6YVMZYLBKG/events.json","paper":"https://pith.science/paper/XA4HPQGD"},"agent_actions":{"view_html":"https://pith.science/pith/XA4HPQGD5W6ZIRDH6YVMZYLBKG","download_json":"https://pith.science/pith/XA4HPQGD5W6ZIRDH6YVMZYLBKG.json","view_paper":"https://pith.science/paper/XA4HPQGD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.08276&json=true","fetch_graph":"https://pith.science/api/pith-number/XA4HPQGD5W6ZIRDH6YVMZYLBKG/graph.json","fetch_events":"https://pith.science/api/pith-number/XA4HPQGD5W6ZIRDH6YVMZYLBKG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XA4HPQGD5W6ZIRDH6YVMZYLBKG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XA4HPQGD5W6ZIRDH6YVMZYLBKG/action/storage_attestation","attest_author":"https://pith.science/pith/XA4HPQGD5W6ZIRDH6YVMZYLBKG/action/author_attestation","sign_citation":"https://pith.science/pith/XA4HPQGD5W6ZIRDH6YVMZYLBKG/action/citation_signature","submit_replication":"https://pith.science/pith/XA4HPQGD5W6ZIRDH6YVMZYLBKG/action/replication_record"}},"created_at":"2026-07-05T04:28:23.240652+00:00","updated_at":"2026-07-05T04:28:23.240652+00:00"}