{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GNEH6KHNO5HKYKXG2BKI3EW2ZT","short_pith_number":"pith:GNEH6KHN","schema_version":"1.0","canonical_sha256":"33487f28ed774eac2ae6d0548d92dacce5e3c3a01d342ceeb58836d537e76f90","source":{"kind":"arxiv","id":"2408.15045","version":3},"attestation_state":"computed","paper":{"title":"DocLayLLM: An Efficient Multi-modal Extension of Large Language Models for Text-rich Document Understanding","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengyu Wang, Hongliang Li, Jiapeng Wang, Jun Huang, Lianwen Jin, Wenhui Liao","submitted_at":"2024-08-27T13:13:38Z","abstract_excerpt":"Text-rich document understanding (TDU) requires comprehensive analysis of documents containing substantial textual content and complex layouts. While Multimodal Large Language Models (MLLMs) have achieved fast progress in this domain, existing approaches either demand significant computational resources or struggle with effective multi-modal integration. In this paper, we introduce DocLayLLM, an efficient multi-modal extension of LLMs specifically designed for TDU. By lightly integrating visual patch tokens and 2D positional tokens into LLMs' input and encoding the document content using the L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.15045","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-27T13:13:38Z","cross_cats_sorted":[],"title_canon_sha256":"1e650ee67a7609f123c2ae73ad996798303d479ae83f9a0f9c5f7f7f9f215a3c","abstract_canon_sha256":"99f16fe5e92fdfcb379352a1ae6f6d41c17f4ee266d4bb7a6a13138470777e9f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:34:09.934047Z","signature_b64":"jeVY4Zw0BgHNIQVDtDp0exH+Rkah5NEkf1H6ndL5Jxdue0RxvuCn2NAeANPpnLtPOTnb2GGkRtxyH0tx/nf6BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"33487f28ed774eac2ae6d0548d92dacce5e3c3a01d342ceeb58836d537e76f90","last_reissued_at":"2026-07-05T10:34:09.933091Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:34:09.933091Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DocLayLLM: An Efficient Multi-modal Extension of Large Language Models for Text-rich Document Understanding","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengyu Wang, Hongliang Li, Jiapeng Wang, Jun Huang, Lianwen Jin, Wenhui Liao","submitted_at":"2024-08-27T13:13:38Z","abstract_excerpt":"Text-rich document understanding (TDU) requires comprehensive analysis of documents containing substantial textual content and complex layouts. While Multimodal Large Language Models (MLLMs) have achieved fast progress in this domain, existing approaches either demand significant computational resources or struggle with effective multi-modal integration. In this paper, we introduce DocLayLLM, an efficient multi-modal extension of LLMs specifically designed for TDU. By lightly integrating visual patch tokens and 2D positional tokens into LLMs' input and encoding the document content using the L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.15045","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.15045/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.15045","created_at":"2026-07-05T10:34:09.933216+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.15045v3","created_at":"2026-07-05T10:34:09.933216+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.15045","created_at":"2026-07-05T10:34:09.933216+00:00"},{"alias_kind":"pith_short_12","alias_value":"GNEH6KHNO5HK","created_at":"2026-07-05T10:34:09.933216+00:00"},{"alias_kind":"pith_short_16","alias_value":"GNEH6KHNO5HKYKXG","created_at":"2026-07-05T10:34:09.933216+00:00"},{"alias_kind":"pith_short_8","alias_value":"GNEH6KHN","created_at":"2026-07-05T10:34:09.933216+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.09861","citing_title":"A Survey on MLLM-based Visually Rich Document Understanding: Methods, Challenges, and Emerging Trends","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00321","citing_title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2511.22521","citing_title":"DocVAL: Validated Chain-of-Thought Distillation for Grounded Document VQA","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00161","citing_title":"Q-Mask: Query-driven Causal Masks for Text Anchoring in OCR-Oriented Vision-Language Models","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GNEH6KHNO5HKYKXG2BKI3EW2ZT","json":"https://pith.science/pith/GNEH6KHNO5HKYKXG2BKI3EW2ZT.json","graph_json":"https://pith.science/api/pith-number/GNEH6KHNO5HKYKXG2BKI3EW2ZT/graph.json","events_json":"https://pith.science/api/pith-number/GNEH6KHNO5HKYKXG2BKI3EW2ZT/events.json","paper":"https://pith.science/paper/GNEH6KHN"},"agent_actions":{"view_html":"https://pith.science/pith/GNEH6KHNO5HKYKXG2BKI3EW2ZT","download_json":"https://pith.science/pith/GNEH6KHNO5HKYKXG2BKI3EW2ZT.json","view_paper":"https://pith.science/paper/GNEH6KHN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.15045&json=true","fetch_graph":"https://pith.science/api/pith-number/GNEH6KHNO5HKYKXG2BKI3EW2ZT/graph.json","fetch_events":"https://pith.science/api/pith-number/GNEH6KHNO5HKYKXG2BKI3EW2ZT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GNEH6KHNO5HKYKXG2BKI3EW2ZT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GNEH6KHNO5HKYKXG2BKI3EW2ZT/action/storage_attestation","attest_author":"https://pith.science/pith/GNEH6KHNO5HKYKXG2BKI3EW2ZT/action/author_attestation","sign_citation":"https://pith.science/pith/GNEH6KHNO5HKYKXG2BKI3EW2ZT/action/citation_signature","submit_replication":"https://pith.science/pith/GNEH6KHNO5HKYKXG2BKI3EW2ZT/action/replication_record"}},"created_at":"2026-07-05T10:34:09.933216+00:00","updated_at":"2026-07-05T10:34:09.933216+00:00"}