{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:REMBTKCIRFERYSZTWHCWLVTUXP","short_pith_number":"pith:REMBTKCI","schema_version":"1.0","canonical_sha256":"891819a84889491c4b33b1c565d674bbead8f517147020ffeb8123bb810e3aa5","source":{"kind":"arxiv","id":"2407.20662","version":1},"attestation_state":"computed","paper":{"title":"DocXPand-25k: a large and diverse benchmark dataset for identity documents analysis","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexis Berg\\`es, Evgeny Stepankevich, Guillaume Betmont, Julien Lerouge, Thomas Bres","submitted_at":"2024-07-30T08:55:27Z","abstract_excerpt":"Identity document (ID) image analysis has become essential for many online services, like bank account opening or insurance subscription. In recent years, much research has been conducted on subjects like document localization, text recognition and fraud detection, to achieve a level of accuracy reliable enough to automatize identity verification. However, there are only a few available datasets to benchmark ID analysis methods, mainly because of privacy restrictions, security requirements and legal reasons.\n  In this paper, we present the DocXPand-25k dataset, which consists of 24,994 richly "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.20662","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-30T08:55:27Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ac70dcfe6571a969c75332726d7c791a218dbf73ea6e4b6bc8a9025b8849188e","abstract_canon_sha256":"3fa19542b8aa416e2ac2019f2721614b6e35cee31667fd6402575aca5d25ea90"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:50:16.077740Z","signature_b64":"gmnDQzvCYJTLP9g2xrpS5l3Ir+lrqJvWq/mqgjHALxOhBqS60DbIqvE43f6jtbtEIFTkCH0ShaGUHw4PCMW2BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"891819a84889491c4b33b1c565d674bbead8f517147020ffeb8123bb810e3aa5","last_reissued_at":"2026-07-05T08:50:16.077406Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:50:16.077406Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DocXPand-25k: a large and diverse benchmark dataset for identity documents analysis","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexis Berg\\`es, Evgeny Stepankevich, Guillaume Betmont, Julien Lerouge, Thomas Bres","submitted_at":"2024-07-30T08:55:27Z","abstract_excerpt":"Identity document (ID) image analysis has become essential for many online services, like bank account opening or insurance subscription. In recent years, much research has been conducted on subjects like document localization, text recognition and fraud detection, to achieve a level of accuracy reliable enough to automatize identity verification. However, there are only a few available datasets to benchmark ID analysis methods, mainly because of privacy restrictions, security requirements and legal reasons.\n  In this paper, we present the DocXPand-25k dataset, which consists of 24,994 richly "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.20662","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.20662/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.20662","created_at":"2026-07-05T08:50:16.077460+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.20662v1","created_at":"2026-07-05T08:50:16.077460+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.20662","created_at":"2026-07-05T08:50:16.077460+00:00"},{"alias_kind":"pith_short_12","alias_value":"REMBTKCIRFER","created_at":"2026-07-05T08:50:16.077460+00:00"},{"alias_kind":"pith_short_16","alias_value":"REMBTKCIRFERYSZT","created_at":"2026-07-05T08:50:16.077460+00:00"},{"alias_kind":"pith_short_8","alias_value":"REMBTKCI","created_at":"2026-07-05T08:50:16.077460+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.12911","citing_title":"Beyond Visual Evidence: Revealing and Mitigating Relational Privacy Leakage in Document MLLMs","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/REMBTKCIRFERYSZTWHCWLVTUXP","json":"https://pith.science/pith/REMBTKCIRFERYSZTWHCWLVTUXP.json","graph_json":"https://pith.science/api/pith-number/REMBTKCIRFERYSZTWHCWLVTUXP/graph.json","events_json":"https://pith.science/api/pith-number/REMBTKCIRFERYSZTWHCWLVTUXP/events.json","paper":"https://pith.science/paper/REMBTKCI"},"agent_actions":{"view_html":"https://pith.science/pith/REMBTKCIRFERYSZTWHCWLVTUXP","download_json":"https://pith.science/pith/REMBTKCIRFERYSZTWHCWLVTUXP.json","view_paper":"https://pith.science/paper/REMBTKCI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.20662&json=true","fetch_graph":"https://pith.science/api/pith-number/REMBTKCIRFERYSZTWHCWLVTUXP/graph.json","fetch_events":"https://pith.science/api/pith-number/REMBTKCIRFERYSZTWHCWLVTUXP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/REMBTKCIRFERYSZTWHCWLVTUXP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/REMBTKCIRFERYSZTWHCWLVTUXP/action/storage_attestation","attest_author":"https://pith.science/pith/REMBTKCIRFERYSZTWHCWLVTUXP/action/author_attestation","sign_citation":"https://pith.science/pith/REMBTKCIRFERYSZTWHCWLVTUXP/action/citation_signature","submit_replication":"https://pith.science/pith/REMBTKCIRFERYSZTWHCWLVTUXP/action/replication_record"}},"created_at":"2026-07-05T08:50:16.077460+00:00","updated_at":"2026-07-05T08:50:16.077460+00:00"}