{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IZR45QEIBM35XBGLYMTM5HGYEJ","short_pith_number":"pith:IZR45QEI","schema_version":"1.0","canonical_sha256":"4663cec0880b37db84cbc326ce9cd8226dc43e7b538aba7bebbb35266ba38b55","source":{"kind":"arxiv","id":"2305.08719","version":2},"attestation_state":"computed","paper":{"title":"M$^{6}$Doc: A Large-Scale Multi-Format, Multi-Type, Multi-Layout, Multi-Language, Multi-Annotation Category Dataset for Modern Document Layout Analysis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hiuyi Cheng, Jiaxin Zhang, Jing Li, Kai Ding, Lianwen Jin, Peirong Zhang, Qiyuan Zhu, Sihang Wu, Zecheng Xie","submitted_at":"2023-05-15T15:29:06Z","abstract_excerpt":"Document layout analysis is a crucial prerequisite for document understanding, including document retrieval and conversion. Most public datasets currently contain only PDF documents and lack realistic documents. Models trained on these datasets may not generalize well to real-world scenarios. Therefore, this paper introduces a large and diverse document layout analysis dataset called $M^{6}Doc$. The $M^6$ designation represents six properties: (1) Multi-Format (including scanned, photographed, and PDF documents); (2) Multi-Type (such as scientific articles, textbooks, books, test papers, magaz"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.08719","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-05-15T15:29:06Z","cross_cats_sorted":[],"title_canon_sha256":"57a87b7fe4324b9f462a23df426466025582dbcbd44f4ce2a9dc8f4bd5591b4e","abstract_canon_sha256":"7ebbd3a6e91c0668c147c436cf37ab85a66c45d025269c30a241069cf6923d75"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:12:02.931628Z","signature_b64":"pN8J0rMuHTcVgZ33JqGfiP6MDEr7PHz74UqR0HLdMh8vSJRgd5eJmkRvQzrWkoz120J2G0XwV7Dw1bGn/Bl9Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4663cec0880b37db84cbc326ce9cd8226dc43e7b538aba7bebbb35266ba38b55","last_reissued_at":"2026-07-05T06:12:02.931282Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:12:02.931282Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"M$^{6}$Doc: A Large-Scale Multi-Format, Multi-Type, Multi-Layout, Multi-Language, Multi-Annotation Category Dataset for Modern Document Layout Analysis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hiuyi Cheng, Jiaxin Zhang, Jing Li, Kai Ding, Lianwen Jin, Peirong Zhang, Qiyuan Zhu, Sihang Wu, Zecheng Xie","submitted_at":"2023-05-15T15:29:06Z","abstract_excerpt":"Document layout analysis is a crucial prerequisite for document understanding, including document retrieval and conversion. Most public datasets currently contain only PDF documents and lack realistic documents. Models trained on these datasets may not generalize well to real-world scenarios. Therefore, this paper introduces a large and diverse document layout analysis dataset called $M^{6}Doc$. The $M^6$ designation represents six properties: (1) Multi-Format (including scanned, photographed, and PDF documents); (2) Multi-Type (such as scientific articles, textbooks, books, test papers, magaz"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.08719","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.08719/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.08719","created_at":"2026-07-05T06:12:02.931343+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.08719v2","created_at":"2026-07-05T06:12:02.931343+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.08719","created_at":"2026-07-05T06:12:02.931343+00:00"},{"alias_kind":"pith_short_12","alias_value":"IZR45QEIBM35","created_at":"2026-07-05T06:12:02.931343+00:00"},{"alias_kind":"pith_short_16","alias_value":"IZR45QEIBM35XBGL","created_at":"2026-07-05T06:12:02.931343+00:00"},{"alias_kind":"pith_short_8","alias_value":"IZR45QEI","created_at":"2026-07-05T06:12:02.931343+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.00321","citing_title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","ref_index":96,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IZR45QEIBM35XBGLYMTM5HGYEJ","json":"https://pith.science/pith/IZR45QEIBM35XBGLYMTM5HGYEJ.json","graph_json":"https://pith.science/api/pith-number/IZR45QEIBM35XBGLYMTM5HGYEJ/graph.json","events_json":"https://pith.science/api/pith-number/IZR45QEIBM35XBGLYMTM5HGYEJ/events.json","paper":"https://pith.science/paper/IZR45QEI"},"agent_actions":{"view_html":"https://pith.science/pith/IZR45QEIBM35XBGLYMTM5HGYEJ","download_json":"https://pith.science/pith/IZR45QEIBM35XBGLYMTM5HGYEJ.json","view_paper":"https://pith.science/paper/IZR45QEI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.08719&json=true","fetch_graph":"https://pith.science/api/pith-number/IZR45QEIBM35XBGLYMTM5HGYEJ/graph.json","fetch_events":"https://pith.science/api/pith-number/IZR45QEIBM35XBGLYMTM5HGYEJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IZR45QEIBM35XBGLYMTM5HGYEJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IZR45QEIBM35XBGLYMTM5HGYEJ/action/storage_attestation","attest_author":"https://pith.science/pith/IZR45QEIBM35XBGLYMTM5HGYEJ/action/author_attestation","sign_citation":"https://pith.science/pith/IZR45QEIBM35XBGLYMTM5HGYEJ/action/citation_signature","submit_replication":"https://pith.science/pith/IZR45QEIBM35XBGLYMTM5HGYEJ/action/replication_record"}},"created_at":"2026-07-05T06:12:02.931343+00:00","updated_at":"2026-07-05T06:12:02.931343+00:00"}