{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K2BIASC7R3XXEVUX45SDRUPYFH","short_pith_number":"pith:K2BIASC7","schema_version":"1.0","canonical_sha256":"568280485f8eef725697e76438d1f829dca4c3a0f31e36a8bffc078ceacd3b57","source":{"kind":"arxiv","id":"2411.10068","version":1},"attestation_state":"computed","paper":{"title":"Diachronic Document Dataset for Semantic Layout Analysis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Beno\\^it Sagot, DCIS), Florian Cafiero (PSL), Hugo Scheithauer, Juliette Janes (ALMAnaCH), Laurent Romary (ALMAnaCH, Sarah B\\'eni\\`ere (ALMAnaCH), Simon Gabay, Thibault Cl\\'erice (ALMAnaCH)","submitted_at":"2024-11-15T09:33:13Z","abstract_excerpt":"We present a novel, open-access dataset designed for semantic layout analysis, built to support document recreation workflows through mapping with the Text Encoding Initiative (TEI) standard. This dataset includes 7,254 annotated pages spanning a large temporal range (1600-2024) of digitised and born-digital materials across diverse document types (magazines, papers from sciences and humanities, PhD theses, monographs, plays, administrative reports, etc.) sorted into modular subsets. By incorporating content from different periods and genres, it addresses varying layout complexities and histor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.10068","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-15T09:33:13Z","cross_cats_sorted":[],"title_canon_sha256":"77408b469af53a757f681f448c57b5e06681b0aa66183915ff5005b0f0f0adec","abstract_canon_sha256":"257e18b9003ddf1e581ca3ba3e38ce640b9703ce96092528e1e4f74ccd3839b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:58.315897Z","signature_b64":"NtOFDCj+m+cmkng/7vXtB/RS8gLonQGdJVna5jXbdJyIyFcBkMNHYUi11y4MEMdYGxuGKwAkalpOKNBiKZyMBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"568280485f8eef725697e76438d1f829dca4c3a0f31e36a8bffc078ceacd3b57","last_reissued_at":"2026-07-05T09:35:58.315282Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:58.315282Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Diachronic Document Dataset for Semantic Layout Analysis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Beno\\^it Sagot, DCIS), Florian Cafiero (PSL), Hugo Scheithauer, Juliette Janes (ALMAnaCH), Laurent Romary (ALMAnaCH, Sarah B\\'eni\\`ere (ALMAnaCH), Simon Gabay, Thibault Cl\\'erice (ALMAnaCH)","submitted_at":"2024-11-15T09:33:13Z","abstract_excerpt":"We present a novel, open-access dataset designed for semantic layout analysis, built to support document recreation workflows through mapping with the Text Encoding Initiative (TEI) standard. This dataset includes 7,254 annotated pages spanning a large temporal range (1600-2024) of digitised and born-digital materials across diverse document types (magazines, papers from sciences and humanities, PhD theses, monographs, plays, administrative reports, etc.) sorted into modular subsets. By incorporating content from different periods and genres, it addresses varying layout complexities and histor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.10068","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.10068/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.10068","created_at":"2026-07-05T09:35:58.315362+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.10068v1","created_at":"2026-07-05T09:35:58.315362+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.10068","created_at":"2026-07-05T09:35:58.315362+00:00"},{"alias_kind":"pith_short_12","alias_value":"K2BIASC7R3XX","created_at":"2026-07-05T09:35:58.315362+00:00"},{"alias_kind":"pith_short_16","alias_value":"K2BIASC7R3XXEVUX","created_at":"2026-07-05T09:35:58.315362+00:00"},{"alias_kind":"pith_short_8","alias_value":"K2BIASC7","created_at":"2026-07-05T09:35:58.315362+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.04424","citing_title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","ref_index":2024,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K2BIASC7R3XXEVUX45SDRUPYFH","json":"https://pith.science/pith/K2BIASC7R3XXEVUX45SDRUPYFH.json","graph_json":"https://pith.science/api/pith-number/K2BIASC7R3XXEVUX45SDRUPYFH/graph.json","events_json":"https://pith.science/api/pith-number/K2BIASC7R3XXEVUX45SDRUPYFH/events.json","paper":"https://pith.science/paper/K2BIASC7"},"agent_actions":{"view_html":"https://pith.science/pith/K2BIASC7R3XXEVUX45SDRUPYFH","download_json":"https://pith.science/pith/K2BIASC7R3XXEVUX45SDRUPYFH.json","view_paper":"https://pith.science/paper/K2BIASC7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.10068&json=true","fetch_graph":"https://pith.science/api/pith-number/K2BIASC7R3XXEVUX45SDRUPYFH/graph.json","fetch_events":"https://pith.science/api/pith-number/K2BIASC7R3XXEVUX45SDRUPYFH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K2BIASC7R3XXEVUX45SDRUPYFH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K2BIASC7R3XXEVUX45SDRUPYFH/action/storage_attestation","attest_author":"https://pith.science/pith/K2BIASC7R3XXEVUX45SDRUPYFH/action/author_attestation","sign_citation":"https://pith.science/pith/K2BIASC7R3XXEVUX45SDRUPYFH/action/citation_signature","submit_replication":"https://pith.science/pith/K2BIASC7R3XXEVUX45SDRUPYFH/action/replication_record"}},"created_at":"2026-07-05T09:35:58.315362+00:00","updated_at":"2026-07-05T09:35:58.315362+00:00"}