{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:5YX7IDEFMUBCJUCC5PUK45Q32P","short_pith_number":"pith:5YX7IDEF","schema_version":"1.0","canonical_sha256":"ee2ff40c85650224d042ebe8ae761bd3d54249a2a24cf6d54eebcbeefe938c6c","source":{"kind":"arxiv","id":"2601.20430","version":2},"attestation_state":"computed","paper":{"title":"Youtu-Parsing: Perception, Structuring and Recognition via High-Parallelism Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Antai Guo, Bing Liu, Chengxu He, Deqiang Jiang, Haodong Lin, Haoyu Cao, Huang Chen, Kun Yin, Qianyu Li, Shuangyin Liu, Xiaotian Li, Xing Sun, Xin Li, Yanqiu Qu, Yanzhen Liao, Yinsong Liu, Yunfei Wu, Yunsheng Wu, Zhongpeng Cai","submitted_at":"2026-01-28T09:37:13Z","abstract_excerpt":"This paper presents Youtu-Parsing, an efficient and versatile document parsing model designed for high-performance content extraction. The architecture employs a native Vision Transformer (ViT) featuring a dynamic-resolution visual encoder to extract shared document features, coupled with a prompt-guided Youtu-LLM-2B language model for layout analysis and region-prompted decoding. Leveraging this decoupled and feature-reusable framework, we introduce a high-parallelism decoding strategy comprising two core components: token parallelism and query parallelism. The token parallelism strategy conc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2601.20430","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-01-28T09:37:13Z","cross_cats_sorted":[],"title_canon_sha256":"9556ae2863ac688678fe40f07925ba5253855a8c3d3e5cce96c9e212d2966c51","abstract_canon_sha256":"5e9a2b5ea37b48410e833d9377e7998ec3b7dcbac3aed4982f1ccdc0f4820c07"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T02:21:28.430186Z","signature_b64":"rq0inLQ+e4d/MxjWzX/360q5eB6HjjJ7ayUQyyjDWHbNY8d0Qz3wV8Ax9hhU4XzLJRqyZHzhMjp+TKJOXVdbCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ee2ff40c85650224d042ebe8ae761bd3d54249a2a24cf6d54eebcbeefe938c6c","last_reissued_at":"2026-07-14T02:21:28.429283Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T02:21:28.429283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Youtu-Parsing: Perception, Structuring and Recognition via High-Parallelism Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Antai Guo, Bing Liu, Chengxu He, Deqiang Jiang, Haodong Lin, Haoyu Cao, Huang Chen, Kun Yin, Qianyu Li, Shuangyin Liu, Xiaotian Li, Xing Sun, Xin Li, Yanqiu Qu, Yanzhen Liao, Yinsong Liu, Yunfei Wu, Yunsheng Wu, Zhongpeng Cai","submitted_at":"2026-01-28T09:37:13Z","abstract_excerpt":"This paper presents Youtu-Parsing, an efficient and versatile document parsing model designed for high-performance content extraction. The architecture employs a native Vision Transformer (ViT) featuring a dynamic-resolution visual encoder to extract shared document features, coupled with a prompt-guided Youtu-LLM-2B language model for layout analysis and region-prompted decoding. Leveraging this decoupled and feature-reusable framework, we introduce a high-parallelism decoding strategy comprising two core components: token parallelism and query parallelism. The token parallelism strategy conc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2601.20430","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2601.20430/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2601.20430","created_at":"2026-07-14T02:21:28.429702+00:00"},{"alias_kind":"arxiv_version","alias_value":"2601.20430v2","created_at":"2026-07-14T02:21:28.429702+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2601.20430","created_at":"2026-07-14T02:21:28.429702+00:00"},{"alias_kind":"pith_short_12","alias_value":"5YX7IDEFMUBC","created_at":"2026-07-14T02:21:28.429702+00:00"},{"alias_kind":"pith_short_16","alias_value":"5YX7IDEFMUBCJUCC","created_at":"2026-07-14T02:21:28.429702+00:00"},{"alias_kind":"pith_short_8","alias_value":"5YX7IDEF","created_at":"2026-07-14T02:21:28.429702+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":8,"sample":[{"citing_arxiv_id":"2606.24447","citing_title":"P-MTP: Efficient Document Parsing via Multi-Token Prediction with Progressive Depth Scaling","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03264","citing_title":"PaddleOCR-VL-1.6: Expanding the Frontier of Document Parsing with Under-Optimized Region Refinement and Progressive Post-Training","ref_index":41,"is_internal_anchor":true},{"citing_arxiv_id":"2605.22100","citing_title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","ref_index":59,"is_internal_anchor":true},{"citing_arxiv_id":"2605.27978","citing_title":"ABot-OCR Technical Report","ref_index":51,"is_internal_anchor":true},{"citing_arxiv_id":"2605.22100","citing_title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","ref_index":59,"is_internal_anchor":true},{"citing_arxiv_id":"2604.02692","citing_title":"Parser-Oriented Structural Refinement for a Stable Layout Interface in Document Parsing","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2605.07492","citing_title":"How Far Is Document Parsing from Solved? PureDocBench: A Source-TraceableBenchmark across Clean, Degraded, and Real-World Settings","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2604.04771","citing_title":"MinerU2.5-Pro: Pushing the Limits of Data-Centric Document Parsing at Scale","ref_index":46,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5YX7IDEFMUBCJUCC5PUK45Q32P","json":"https://pith.science/pith/5YX7IDEFMUBCJUCC5PUK45Q32P.json","graph_json":"https://pith.science/api/pith-number/5YX7IDEFMUBCJUCC5PUK45Q32P/graph.json","events_json":"https://pith.science/api/pith-number/5YX7IDEFMUBCJUCC5PUK45Q32P/events.json","paper":"https://pith.science/paper/5YX7IDEF"},"agent_actions":{"view_html":"https://pith.science/pith/5YX7IDEFMUBCJUCC5PUK45Q32P","download_json":"https://pith.science/pith/5YX7IDEFMUBCJUCC5PUK45Q32P.json","view_paper":"https://pith.science/paper/5YX7IDEF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2601.20430&json=true","fetch_graph":"https://pith.science/api/pith-number/5YX7IDEFMUBCJUCC5PUK45Q32P/graph.json","fetch_events":"https://pith.science/api/pith-number/5YX7IDEFMUBCJUCC5PUK45Q32P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5YX7IDEFMUBCJUCC5PUK45Q32P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5YX7IDEFMUBCJUCC5PUK45Q32P/action/storage_attestation","attest_author":"https://pith.science/pith/5YX7IDEFMUBCJUCC5PUK45Q32P/action/author_attestation","sign_citation":"https://pith.science/pith/5YX7IDEFMUBCJUCC5PUK45Q32P/action/citation_signature","submit_replication":"https://pith.science/pith/5YX7IDEFMUBCJUCC5PUK45Q32P/action/replication_record"}},"created_at":"2026-07-14T02:21:28.429702+00:00","updated_at":"2026-07-14T02:21:28.429702+00:00"}