{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7GLS5HEJYUXBUR72E4A76K7HRW","short_pith_number":"pith:7GLS5HEJ","schema_version":"1.0","canonical_sha256":"f9972e9c89c52e1a47fa2701ff2be78d8d2bc8b3599bed66ddc22723b970d4da","source":{"kind":"arxiv","id":"2407.01976","version":3},"attestation_state":"computed","paper":{"title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.MM"],"primary_cat":"cs.CL","authors_text":"Binghong Wu, Can Huang, Haiyang Yu, Han Wang, Hao Feng, Hao Liu, Jinghui Lu, Jingqun Tang, Qi Liu, Yanjie Wang, YongJie Ye, Ziwei Yang","submitted_at":"2024-07-02T06:29:05Z","abstract_excerpt":"Recently, many studies have demonstrated that exclusively incorporating OCR-derived text and spatial layouts with large language models (LLMs) can be highly effective for document understanding tasks. However, existing methods that integrate spatial layouts with text have limitations, such as producing overly long text sequences or failing to fully leverage the autoregressive traits of LLMs. In this work, we introduce Interleaving Layout and Text in a Large Language Model (LayTextLLM)} for document understanding. LayTextLLM projects each bounding box to a single embedding and interleaves it wi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.01976","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-02T06:29:05Z","cross_cats_sorted":["cs.AI","cs.MM"],"title_canon_sha256":"4d3ab0b080c3578ae5fff5b9ba9f760a0efd44ba3651f60f6612a50a98f8e06c","abstract_canon_sha256":"978a2bd2bf6617b11df644b951c9cd01cafd4afa6a2712cdd4fff2354dac0d3d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:38.207377Z","signature_b64":"Yg3vC0AII6Z5eY4NB4A/6T7dAeQX7ji8ml/HBOMECSMPYaRpBlFW6q8S0bYWHA/ClYp55oWQZqknQh/qjvf1Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f9972e9c89c52e1a47fa2701ff2be78d8d2bc8b3599bed66ddc22723b970d4da","last_reissued_at":"2026-07-05T11:04:38.206878Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:38.206878Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.MM"],"primary_cat":"cs.CL","authors_text":"Binghong Wu, Can Huang, Haiyang Yu, Han Wang, Hao Feng, Hao Liu, Jinghui Lu, Jingqun Tang, Qi Liu, Yanjie Wang, YongJie Ye, Ziwei Yang","submitted_at":"2024-07-02T06:29:05Z","abstract_excerpt":"Recently, many studies have demonstrated that exclusively incorporating OCR-derived text and spatial layouts with large language models (LLMs) can be highly effective for document understanding tasks. However, existing methods that integrate spatial layouts with text have limitations, such as producing overly long text sequences or failing to fully leverage the autoregressive traits of LLMs. In this work, we introduce Interleaving Layout and Text in a Large Language Model (LayTextLLM)} for document understanding. LayTextLLM projects each bounding box to a single embedding and interleaves it wi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.01976","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.01976/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.01976","created_at":"2026-07-05T11:04:38.206937+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.01976v3","created_at":"2026-07-05T11:04:38.206937+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.01976","created_at":"2026-07-05T11:04:38.206937+00:00"},{"alias_kind":"pith_short_12","alias_value":"7GLS5HEJYUXB","created_at":"2026-07-05T11:04:38.206937+00:00"},{"alias_kind":"pith_short_16","alias_value":"7GLS5HEJYUXBUR72","created_at":"2026-07-05T11:04:38.206937+00:00"},{"alias_kind":"pith_short_8","alias_value":"7GLS5HEJ","created_at":"2026-07-05T11:04:38.206937+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01602","citing_title":"ProWAFT: A ROMA-LPD Instance for Workload-Aware and Dynamic Fault Tolerance in FPGA-Based CNN Accelerators","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30189","citing_title":"DAIN: Dynamic Agent-Based Interaction Network for Efficient and Collaborative Multimodal Reasoning","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2507.09861","citing_title":"A Survey on MLLM-based Visually Rich Document Understanding: Methods, Challenges, and Emerging Trends","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2511.22521","citing_title":"DocVAL: Validated Chain-of-Thought Distillation for Grounded Document VQA","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00161","citing_title":"Q-Mask: Query-driven Causal Masks for Text Anchoring in OCR-Oriented Vision-Language Models","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7GLS5HEJYUXBUR72E4A76K7HRW","json":"https://pith.science/pith/7GLS5HEJYUXBUR72E4A76K7HRW.json","graph_json":"https://pith.science/api/pith-number/7GLS5HEJYUXBUR72E4A76K7HRW/graph.json","events_json":"https://pith.science/api/pith-number/7GLS5HEJYUXBUR72E4A76K7HRW/events.json","paper":"https://pith.science/paper/7GLS5HEJ"},"agent_actions":{"view_html":"https://pith.science/pith/7GLS5HEJYUXBUR72E4A76K7HRW","download_json":"https://pith.science/pith/7GLS5HEJYUXBUR72E4A76K7HRW.json","view_paper":"https://pith.science/paper/7GLS5HEJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.01976&json=true","fetch_graph":"https://pith.science/api/pith-number/7GLS5HEJYUXBUR72E4A76K7HRW/graph.json","fetch_events":"https://pith.science/api/pith-number/7GLS5HEJYUXBUR72E4A76K7HRW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7GLS5HEJYUXBUR72E4A76K7HRW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7GLS5HEJYUXBUR72E4A76K7HRW/action/storage_attestation","attest_author":"https://pith.science/pith/7GLS5HEJYUXBUR72E4A76K7HRW/action/author_attestation","sign_citation":"https://pith.science/pith/7GLS5HEJYUXBUR72E4A76K7HRW/action/citation_signature","submit_replication":"https://pith.science/pith/7GLS5HEJYUXBUR72E4A76K7HRW/action/replication_record"}},"created_at":"2026-07-05T11:04:38.206937+00:00","updated_at":"2026-07-05T11:04:38.206937+00:00"}