{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7HLNSMSKBI3UEKBLXUXQJL5KPG","short_pith_number":"pith:7HLNSMSK","schema_version":"1.0","canonical_sha256":"f9d6d9324a0a3742282bbd2f04afaa79a2369d919a72d7a12014fd77f3377da5","source":{"kind":"arxiv","id":"2412.04262","version":1},"attestation_state":"computed","paper":{"title":"SynFinTabs: A Dataset of Synthetic Financial Tables for Information and Table Extraction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Barry Devereux, Ethan Bradley, Karen Rafferty, Muhammad Roman","submitted_at":"2024-12-05T15:42:59Z","abstract_excerpt":"Table extraction from document images is a challenging AI problem, and labelled data for many content domains is difficult to come by. Existing table extraction datasets often focus on scientific tables due to the vast amount of academic articles that are readily available, along with their source code. However, there are significant layout and typographical differences between tables found across scientific, financial, and other domains. Current datasets often lack the words, and their positions, contained within the tables, instead relying on unreliable OCR to extract these features for trai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.04262","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-05T15:42:59Z","cross_cats_sorted":[],"title_canon_sha256":"9a294b8acadf1b5054e31e4139d631d21726aa10b07e5842e7d38132ca044c48","abstract_canon_sha256":"65e6e58c768535be8103d3ab160260b7f2bc9578b11d49985049d72a13fa62e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:01.689525Z","signature_b64":"JZp+hYI2nMJfEvGRBj+oMZeEp4pahh4FViEt+xAc6Gxt6ecHhh+KmeWaw9WvuKnQKlohnby94nNgIqLSS4XiDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f9d6d9324a0a3742282bbd2f04afaa79a2369d919a72d7a12014fd77f3377da5","last_reissued_at":"2026-07-05T09:45:01.689068Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:01.689068Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SynFinTabs: A Dataset of Synthetic Financial Tables for Information and Table Extraction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Barry Devereux, Ethan Bradley, Karen Rafferty, Muhammad Roman","submitted_at":"2024-12-05T15:42:59Z","abstract_excerpt":"Table extraction from document images is a challenging AI problem, and labelled data for many content domains is difficult to come by. Existing table extraction datasets often focus on scientific tables due to the vast amount of academic articles that are readily available, along with their source code. However, there are significant layout and typographical differences between tables found across scientific, financial, and other domains. Current datasets often lack the words, and their positions, contained within the tables, instead relying on unreliable OCR to extract these features for trai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.04262","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.04262/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.04262","created_at":"2026-07-05T09:45:01.689125+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.04262v1","created_at":"2026-07-05T09:45:01.689125+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.04262","created_at":"2026-07-05T09:45:01.689125+00:00"},{"alias_kind":"pith_short_12","alias_value":"7HLNSMSKBI3U","created_at":"2026-07-05T09:45:01.689125+00:00"},{"alias_kind":"pith_short_16","alias_value":"7HLNSMSKBI3UEKBL","created_at":"2026-07-05T09:45:01.689125+00:00"},{"alias_kind":"pith_short_8","alias_value":"7HLNSMSK","created_at":"2026-07-05T09:45:01.689125+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7HLNSMSKBI3UEKBLXUXQJL5KPG","json":"https://pith.science/pith/7HLNSMSKBI3UEKBLXUXQJL5KPG.json","graph_json":"https://pith.science/api/pith-number/7HLNSMSKBI3UEKBLXUXQJL5KPG/graph.json","events_json":"https://pith.science/api/pith-number/7HLNSMSKBI3UEKBLXUXQJL5KPG/events.json","paper":"https://pith.science/paper/7HLNSMSK"},"agent_actions":{"view_html":"https://pith.science/pith/7HLNSMSKBI3UEKBLXUXQJL5KPG","download_json":"https://pith.science/pith/7HLNSMSKBI3UEKBLXUXQJL5KPG.json","view_paper":"https://pith.science/paper/7HLNSMSK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.04262&json=true","fetch_graph":"https://pith.science/api/pith-number/7HLNSMSKBI3UEKBLXUXQJL5KPG/graph.json","fetch_events":"https://pith.science/api/pith-number/7HLNSMSKBI3UEKBLXUXQJL5KPG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7HLNSMSKBI3UEKBLXUXQJL5KPG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7HLNSMSKBI3UEKBLXUXQJL5KPG/action/storage_attestation","attest_author":"https://pith.science/pith/7HLNSMSKBI3UEKBLXUXQJL5KPG/action/author_attestation","sign_citation":"https://pith.science/pith/7HLNSMSKBI3UEKBLXUXQJL5KPG/action/citation_signature","submit_replication":"https://pith.science/pith/7HLNSMSKBI3UEKBLXUXQJL5KPG/action/replication_record"}},"created_at":"2026-07-05T09:45:01.689125+00:00","updated_at":"2026-07-05T09:45:01.689125+00:00"}