{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GMX2V3O7T4WRQOHK7CRNIXZRTT","short_pith_number":"pith:GMX2V3O7","schema_version":"1.0","canonical_sha256":"332faaeddf9f2d1838eaf8a2d45f319cf0f671b39d7d9dfc2e54589f0e1356df","source":{"kind":"arxiv","id":"2410.12628","version":1},"attestation_state":"computed","paper":{"title":"DocLayout-YOLO: Enhancing Document Layout Analysis through Diverse Synthetic Data and Global-to-Local Adaptive Perception","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Wang, Conghui He, Hengrui Kang, Zhiyuan Zhao","submitted_at":"2024-10-16T14:50:47Z","abstract_excerpt":"Document Layout Analysis is crucial for real-world document understanding systems, but it encounters a challenging trade-off between speed and accuracy: multimodal methods leveraging both text and visual features achieve higher accuracy but suffer from significant latency, whereas unimodal methods relying solely on visual features offer faster processing speeds at the expense of accuracy. To address this dilemma, we introduce DocLayout-YOLO, a novel approach that enhances accuracy while maintaining speed advantages through document-specific optimizations in both pre-training and model design. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12628","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-16T14:50:47Z","cross_cats_sorted":[],"title_canon_sha256":"d21656cc4fb959e7702de2298d8c011b2b498e33be3977fa766807a96ea409bc","abstract_canon_sha256":"cc6fa7f1f1472d4cdab481ecc5ae8b7df41aa498b7b9e438cbc5e5a295e079eb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:31.729823Z","signature_b64":"HWkTvH0w/sHXBrRfzvlo0gOI7qDhJ6uHpdtPkBA43uEi6AzpneF7drap2l+HOdPkAOQpXGDA/gwPDJrvjFTcBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"332faaeddf9f2d1838eaf8a2d45f319cf0f671b39d7d9dfc2e54589f0e1356df","last_reissued_at":"2026-07-05T09:21:31.729401Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:31.729401Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DocLayout-YOLO: Enhancing Document Layout Analysis through Diverse Synthetic Data and Global-to-Local Adaptive Perception","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Wang, Conghui He, Hengrui Kang, Zhiyuan Zhao","submitted_at":"2024-10-16T14:50:47Z","abstract_excerpt":"Document Layout Analysis is crucial for real-world document understanding systems, but it encounters a challenging trade-off between speed and accuracy: multimodal methods leveraging both text and visual features achieve higher accuracy but suffer from significant latency, whereas unimodal methods relying solely on visual features offer faster processing speeds at the expense of accuracy. To address this dilemma, we introduce DocLayout-YOLO, a novel approach that enhances accuracy while maintaining speed advantages through document-specific optimizations in both pre-training and model design. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12628","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12628/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12628","created_at":"2026-07-05T09:21:31.729459+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12628v1","created_at":"2026-07-05T09:21:31.729459+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12628","created_at":"2026-07-05T09:21:31.729459+00:00"},{"alias_kind":"pith_short_12","alias_value":"GMX2V3O7T4WR","created_at":"2026-07-05T09:21:31.729459+00:00"},{"alias_kind":"pith_short_16","alias_value":"GMX2V3O7T4WRQOHK","created_at":"2026-07-05T09:21:31.729459+00:00"},{"alias_kind":"pith_short_8","alias_value":"GMX2V3O7","created_at":"2026-07-05T09:21:31.729459+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07836","citing_title":"Infinity-Parser2 Technical Report","ref_index":78,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23344","citing_title":"RT-DocLayout: Real-Time End-to-End Document Layout Analysis with Reading Order in the Wild","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00596","citing_title":"Semantic-Guided Reading Order Reconstruction in Historical Armenian Newspapers with LLMs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06242","citing_title":"Benchmarking Open-Source Layout Detection Models for Data Snapshot Extraction from Institutional Documents","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27978","citing_title":"ABot-OCR Technical Report","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22829","citing_title":"LFRAG: Layout-oriented Fine-grained Retrieval-Augmented Generation on Multimodal Document Understanding","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17159","citing_title":"MADP: A Multi-Agent Pipeline for Sustainable Document Processing with Human-in-the-Loop","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2508.21720","citing_title":"PosterForest: Hierarchical Multi-Agent Collaboration for Scientific Poster Generation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22186","citing_title":"MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02692","citing_title":"Parser-Oriented Structural Refinement for a Stable Layout Interface in Document Parsing","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10845","citing_title":"BabelDOC: Better Layout-Preserving PDF Translation via Intermediate Representation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10341","citing_title":"PaperFit: Vision-in-the-Loop Typesetting Optimization for Scientific Documents","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11042","citing_title":"Improving Layout Representation Learning Across Inconsistently Annotated Datasets via Agentic Harmonization","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04771","citing_title":"MinerU2.5-Pro: Pushing the Limits of Data-Centric Document Parsing at Scale","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06160","citing_title":"The Character Error Vector: Decomposable errors for page-level OCR evaluation","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GMX2V3O7T4WRQOHK7CRNIXZRTT","json":"https://pith.science/pith/GMX2V3O7T4WRQOHK7CRNIXZRTT.json","graph_json":"https://pith.science/api/pith-number/GMX2V3O7T4WRQOHK7CRNIXZRTT/graph.json","events_json":"https://pith.science/api/pith-number/GMX2V3O7T4WRQOHK7CRNIXZRTT/events.json","paper":"https://pith.science/paper/GMX2V3O7"},"agent_actions":{"view_html":"https://pith.science/pith/GMX2V3O7T4WRQOHK7CRNIXZRTT","download_json":"https://pith.science/pith/GMX2V3O7T4WRQOHK7CRNIXZRTT.json","view_paper":"https://pith.science/paper/GMX2V3O7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12628&json=true","fetch_graph":"https://pith.science/api/pith-number/GMX2V3O7T4WRQOHK7CRNIXZRTT/graph.json","fetch_events":"https://pith.science/api/pith-number/GMX2V3O7T4WRQOHK7CRNIXZRTT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GMX2V3O7T4WRQOHK7CRNIXZRTT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GMX2V3O7T4WRQOHK7CRNIXZRTT/action/storage_attestation","attest_author":"https://pith.science/pith/GMX2V3O7T4WRQOHK7CRNIXZRTT/action/author_attestation","sign_citation":"https://pith.science/pith/GMX2V3O7T4WRQOHK7CRNIXZRTT/action/citation_signature","submit_replication":"https://pith.science/pith/GMX2V3O7T4WRQOHK7CRNIXZRTT/action/replication_record"}},"created_at":"2026-07-05T09:21:31.729459+00:00","updated_at":"2026-07-05T09:21:31.729459+00:00"}