{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J3UTHXGXIPBRRVDZ6NRAL2HMBA","short_pith_number":"pith:J3UTHXGX","schema_version":"1.0","canonical_sha256":"4ee933dcd743c318d479f36205e8ec0818accda86b11dcd4826b4f6a95623e21","source":{"kind":"arxiv","id":"2501.00958","version":4},"attestation_state":"computed","paper":{"title":"2.5 Years in Class: A Multimodal Textbook for Vision-Language Pretraining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Deli Zhao, Hang Zhang, Jiashuo Sun, Lidong Bing, Weiming Lu, Wenqi Zhang, Xin Li, Yongliang Shen, Yueting Zhuang","submitted_at":"2025-01-01T21:29:37Z","abstract_excerpt":"Compared to image-text pair data, interleaved corpora enable Vision-Language Models (VLMs) to understand the world more naturally like humans. However, such existing datasets are crawled from webpage, facing challenges like low knowledge density, loose image-text relations, and poor logical coherence between images. On the other hand, the internet hosts vast instructional videos (e.g., online geometry courses) that are widely used by humans to learn foundational subjects, yet these valuable resources remain underexplored in VLM training. In this paper, we introduce a high-quality \\textbf{multi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.00958","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-01T21:29:37Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"e0e66ed456f82ccd81bbbdae36e536f62fcd3bf681922ae0547ae2b8db9b8ec8","abstract_canon_sha256":"1e36ecab0a7f392422580030843b30120c70b17b12af55e54d49b903f8ab5822"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:16.027673Z","signature_b64":"VdwqVc/JNmmBFazB7/87hv13fBzivs8DF+aJO5i9NS5oe/Nu0kNeotlJ98OvLuNEQHhNN7gDgx/5Hb+VR7EXBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ee933dcd743c318d479f36205e8ec0818accda86b11dcd4826b4f6a95623e21","last_reissued_at":"2026-07-05T11:02:16.027235Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:16.027235Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"2.5 Years in Class: A Multimodal Textbook for Vision-Language Pretraining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Deli Zhao, Hang Zhang, Jiashuo Sun, Lidong Bing, Weiming Lu, Wenqi Zhang, Xin Li, Yongliang Shen, Yueting Zhuang","submitted_at":"2025-01-01T21:29:37Z","abstract_excerpt":"Compared to image-text pair data, interleaved corpora enable Vision-Language Models (VLMs) to understand the world more naturally like humans. However, such existing datasets are crawled from webpage, facing challenges like low knowledge density, loose image-text relations, and poor logical coherence between images. On the other hand, the internet hosts vast instructional videos (e.g., online geometry courses) that are widely used by humans to learn foundational subjects, yet these valuable resources remain underexplored in VLM training. In this paper, we introduce a high-quality \\textbf{multi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.00958","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.00958/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.00958","created_at":"2026-07-05T11:02:16.027289+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.00958v4","created_at":"2026-07-05T11:02:16.027289+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.00958","created_at":"2026-07-05T11:02:16.027289+00:00"},{"alias_kind":"pith_short_12","alias_value":"J3UTHXGXIPBR","created_at":"2026-07-05T11:02:16.027289+00:00"},{"alias_kind":"pith_short_16","alias_value":"J3UTHXGXIPBRRVDZ","created_at":"2026-07-05T11:02:16.027289+00:00"},{"alias_kind":"pith_short_8","alias_value":"J3UTHXGX","created_at":"2026-07-05T11:02:16.027289+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.20670","citing_title":"MMSearch-R1: Incentivizing LMMs to Search","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09677","citing_title":"Logics-Parsing-Omni Technical Report","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09271","citing_title":"Shaping Schema via Language Representation as the Next Frontier for LLM Intelligence Expanding","ref_index":101,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J3UTHXGXIPBRRVDZ6NRAL2HMBA","json":"https://pith.science/pith/J3UTHXGXIPBRRVDZ6NRAL2HMBA.json","graph_json":"https://pith.science/api/pith-number/J3UTHXGXIPBRRVDZ6NRAL2HMBA/graph.json","events_json":"https://pith.science/api/pith-number/J3UTHXGXIPBRRVDZ6NRAL2HMBA/events.json","paper":"https://pith.science/paper/J3UTHXGX"},"agent_actions":{"view_html":"https://pith.science/pith/J3UTHXGXIPBRRVDZ6NRAL2HMBA","download_json":"https://pith.science/pith/J3UTHXGXIPBRRVDZ6NRAL2HMBA.json","view_paper":"https://pith.science/paper/J3UTHXGX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.00958&json=true","fetch_graph":"https://pith.science/api/pith-number/J3UTHXGXIPBRRVDZ6NRAL2HMBA/graph.json","fetch_events":"https://pith.science/api/pith-number/J3UTHXGXIPBRRVDZ6NRAL2HMBA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J3UTHXGXIPBRRVDZ6NRAL2HMBA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J3UTHXGXIPBRRVDZ6NRAL2HMBA/action/storage_attestation","attest_author":"https://pith.science/pith/J3UTHXGXIPBRRVDZ6NRAL2HMBA/action/author_attestation","sign_citation":"https://pith.science/pith/J3UTHXGXIPBRRVDZ6NRAL2HMBA/action/citation_signature","submit_replication":"https://pith.science/pith/J3UTHXGXIPBRRVDZ6NRAL2HMBA/action/replication_record"}},"created_at":"2026-07-05T11:02:16.027289+00:00","updated_at":"2026-07-05T11:02:16.027289+00:00"}