{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NDY2FZUNFRY36DVNYA5KHH6EYK","short_pith_number":"pith:NDY2FZUN","schema_version":"1.0","canonical_sha256":"68f1a2e68d2c71bf0eadc03aa39fc4c2b363d734550636e57e835bb829466100","source":{"kind":"arxiv","id":"2404.12803","version":3},"attestation_state":"computed","paper":{"title":"TextSquare: Scaling up Text-Centric Visual Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Binghong Wu, Can Huang, Chunhui Lin, Hao Feng, Hao Liu, Jingqun Tang, Kuan Lu, Lei Liao, Qi Liu, Shu Wei, Siqi Wang, Wei Shi, Xiang Bai, Yangfan He, Yang Li, Yuan Xie, Yuliang Liu, Zhen Zhao","submitted_at":"2024-04-19T11:38:08Z","abstract_excerpt":"Text-centric visual question answering (VQA) has made great strides with the development of Multimodal Large Language Models (MLLMs), yet open-source models still fall short of leading models like GPT4V and Gemini, partly due to a lack of extensive, high-quality instruction tuning data. To this end, we introduce a new approach for creating a massive, high-quality instruction-tuning dataset, Square-10M, which is generated using closed-source MLLMs. The data construction process, termed Square, consists of four steps: Self-Questioning, Answering, Reasoning, and Evaluation. Our experiments with S"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.12803","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-19T11:38:08Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"bb35a22a3703a26447d479b6c56b02921b27a7a7829bb63accf918bd0f98adf3","abstract_canon_sha256":"986188194bc553236b21edf4a285ea38e9bd45897a6899fb2928830ad6addeb5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:19.057862Z","signature_b64":"Ktk4xba9fKdx9pZN5td2NpkLDmSMu+GJWS3cw1Yc11FJeyhCFCZsxyCY2yC3XkiB3zjZpK3n/o5+ZBUSSNPzCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"68f1a2e68d2c71bf0eadc03aa39fc4c2b363d734550636e57e835bb829466100","last_reissued_at":"2026-07-05T11:19:19.057334Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:19.057334Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TextSquare: Scaling up Text-Centric Visual Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Binghong Wu, Can Huang, Chunhui Lin, Hao Feng, Hao Liu, Jingqun Tang, Kuan Lu, Lei Liao, Qi Liu, Shu Wei, Siqi Wang, Wei Shi, Xiang Bai, Yangfan He, Yang Li, Yuan Xie, Yuliang Liu, Zhen Zhao","submitted_at":"2024-04-19T11:38:08Z","abstract_excerpt":"Text-centric visual question answering (VQA) has made great strides with the development of Multimodal Large Language Models (MLLMs), yet open-source models still fall short of leading models like GPT4V and Gemini, partly due to a lack of extensive, high-quality instruction tuning data. To this end, we introduce a new approach for creating a massive, high-quality instruction-tuning dataset, Square-10M, which is generated using closed-source MLLMs. The data construction process, termed Square, consists of four steps: Self-Questioning, Answering, Reasoning, and Evaluation. Our experiments with S"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.12803","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.12803/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.12803","created_at":"2026-07-05T11:19:19.057397+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.12803v3","created_at":"2026-07-05T11:19:19.057397+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.12803","created_at":"2026-07-05T11:19:19.057397+00:00"},{"alias_kind":"pith_short_12","alias_value":"NDY2FZUNFRY3","created_at":"2026-07-05T11:19:19.057397+00:00"},{"alias_kind":"pith_short_16","alias_value":"NDY2FZUNFRY36DVN","created_at":"2026-07-05T11:19:19.057397+00:00"},{"alias_kind":"pith_short_8","alias_value":"NDY2FZUN","created_at":"2026-07-05T11:19:19.057397+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01602","citing_title":"ProWAFT: A ROMA-LPD Instance for Workload-Aware and Dynamic Fault Tolerance in FPGA-Based CNN Accelerators","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29805","citing_title":"Clearer Sight, Fewer Lies: Oriented Pickup Preference Optimization for Multimodal Hallucination Mitigation","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29805","citing_title":"Clearer Sight, Fewer Lies: Oriented Pickup Preference Optimization for Multimodal Hallucination Mitigation","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30189","citing_title":"DAIN: Dynamic Agent-Based Interaction Network for Efficient and Collaborative Multimodal Reasoning","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18173","citing_title":"Do You Need Text Rectification? Soft Attention Mask Embedding for Rectification-Free Scene Text Spotting","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14548","citing_title":"Local Spatiotemporal Convolutional Network for Robust Gait Recognition","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03339","citing_title":"Hierarchical Awareness Adapters with Hybrid Pyramid Feature Fusion for Dense Depth Prediction","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27353","citing_title":"Gait Recognition via Deep Residual Networks and Multi-Branch Feature Fusion","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25188","citing_title":"Image Classification via Random Dilated Convolution with Multi-Branch Feature Extraction and Context Excitation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25178","citing_title":"Lightweight Real-Time Rendering Parameter Optimization via XGBoost-Driven Lookup Tables","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00885","citing_title":"Multi-Branch Non-Homogeneous Image Dehazing via Concentration Partitioning and Image Fusion","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19233","citing_title":"Adaptive Slicing-Assisted Hyper Inference for Enhanced Small Object Detection in High-Resolution Imagery","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19259","citing_title":"Feature Perturbation Pool-based Fusion Network for Unified Multi-Class Industrial Defect Detection","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NDY2FZUNFRY36DVNYA5KHH6EYK","json":"https://pith.science/pith/NDY2FZUNFRY36DVNYA5KHH6EYK.json","graph_json":"https://pith.science/api/pith-number/NDY2FZUNFRY36DVNYA5KHH6EYK/graph.json","events_json":"https://pith.science/api/pith-number/NDY2FZUNFRY36DVNYA5KHH6EYK/events.json","paper":"https://pith.science/paper/NDY2FZUN"},"agent_actions":{"view_html":"https://pith.science/pith/NDY2FZUNFRY36DVNYA5KHH6EYK","download_json":"https://pith.science/pith/NDY2FZUNFRY36DVNYA5KHH6EYK.json","view_paper":"https://pith.science/paper/NDY2FZUN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.12803&json=true","fetch_graph":"https://pith.science/api/pith-number/NDY2FZUNFRY36DVNYA5KHH6EYK/graph.json","fetch_events":"https://pith.science/api/pith-number/NDY2FZUNFRY36DVNYA5KHH6EYK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NDY2FZUNFRY36DVNYA5KHH6EYK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NDY2FZUNFRY36DVNYA5KHH6EYK/action/storage_attestation","attest_author":"https://pith.science/pith/NDY2FZUNFRY36DVNYA5KHH6EYK/action/author_attestation","sign_citation":"https://pith.science/pith/NDY2FZUNFRY36DVNYA5KHH6EYK/action/citation_signature","submit_replication":"https://pith.science/pith/NDY2FZUNFRY36DVNYA5KHH6EYK/action/replication_record"}},"created_at":"2026-07-05T11:19:19.057397+00:00","updated_at":"2026-07-05T11:19:19.057397+00:00"}