{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:63BVLHCD6J25ADKYGTBKPALHDM","short_pith_number":"pith:63BVLHCD","schema_version":"1.0","canonical_sha256":"f6c3559c43f275d00d5834c2a781671b24911bc7823407dcb4b6d790cbe4dbcd","source":{"kind":"arxiv","id":"2410.09871","version":2},"attestation_state":"computed","paper":{"title":"A Comparative Study of PDF Parsing Tools Across Diverse Document Categories","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DL"],"primary_cat":"cs.IR","authors_text":"Narayan S. Adhikari, Shradha Agarwal","submitted_at":"2024-10-13T15:11:31Z","abstract_excerpt":"PDF is one of the most prominent data formats, making PDF parsing crucial for information extraction and retrieval, particularly with the rise of RAG systems. While various PDF parsing tools exist, their effectiveness across different document types remains understudied, especially beyond academic papers. Our research aims to address this gap by comparing 10 popular PDF parsing tools across 6 document categories using the DocLayNet dataset. These tools include PyPDF, pdfminer-six, PyMuPDF, pdfplumber, pypdfium2, Unstructured, Tabula, Camelot, as well as the deep learning-based tools Nougat and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09871","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2024-10-13T15:11:31Z","cross_cats_sorted":["cs.DL"],"title_canon_sha256":"8df3428527f52a5549c1e27e595fa88c6b058f9036f66b917d86658fcaf0b2f6","abstract_canon_sha256":"abc6838b8854b27abb4bc1cdcdc5687f51c5f37eb5b385256f789f483aac42e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:40.140891Z","signature_b64":"+APUzZ3kAZNTe0XRSor7nUWEGRNKJSNWJ/69pRblCJeqOLsSA6+DUvL0VLXEU4ojHLuX0QRbulbgIaMGmgehBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f6c3559c43f275d00d5834c2a781671b24911bc7823407dcb4b6d790cbe4dbcd","last_reissued_at":"2026-07-05T10:43:40.140431Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:40.140431Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comparative Study of PDF Parsing Tools Across Diverse Document Categories","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DL"],"primary_cat":"cs.IR","authors_text":"Narayan S. Adhikari, Shradha Agarwal","submitted_at":"2024-10-13T15:11:31Z","abstract_excerpt":"PDF is one of the most prominent data formats, making PDF parsing crucial for information extraction and retrieval, particularly with the rise of RAG systems. While various PDF parsing tools exist, their effectiveness across different document types remains understudied, especially beyond academic papers. Our research aims to address this gap by comparing 10 popular PDF parsing tools across 6 document categories using the DocLayNet dataset. These tools include PyPDF, pdfminer-six, PyMuPDF, pdfplumber, pypdfium2, Unstructured, Tabula, Camelot, as well as the deep learning-based tools Nougat and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09871","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09871/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09871","created_at":"2026-07-05T10:43:40.140488+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09871v2","created_at":"2026-07-05T10:43:40.140488+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09871","created_at":"2026-07-05T10:43:40.140488+00:00"},{"alias_kind":"pith_short_12","alias_value":"63BVLHCD6J25","created_at":"2026-07-05T10:43:40.140488+00:00"},{"alias_kind":"pith_short_16","alias_value":"63BVLHCD6J25ADKY","created_at":"2026-07-05T10:43:40.140488+00:00"},{"alias_kind":"pith_short_8","alias_value":"63BVLHCD","created_at":"2026-07-05T10:43:40.140488+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.00003","citing_title":"Tabular PDF Information Extraction with Local LLMs and Layout-Aware Parsing: A Reliability Evaluation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12047","citing_title":"Empirical Evaluation of PDF Parsing and Chunking for Financial Question Answering with RAG","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/63BVLHCD6J25ADKYGTBKPALHDM","json":"https://pith.science/pith/63BVLHCD6J25ADKYGTBKPALHDM.json","graph_json":"https://pith.science/api/pith-number/63BVLHCD6J25ADKYGTBKPALHDM/graph.json","events_json":"https://pith.science/api/pith-number/63BVLHCD6J25ADKYGTBKPALHDM/events.json","paper":"https://pith.science/paper/63BVLHCD"},"agent_actions":{"view_html":"https://pith.science/pith/63BVLHCD6J25ADKYGTBKPALHDM","download_json":"https://pith.science/pith/63BVLHCD6J25ADKYGTBKPALHDM.json","view_paper":"https://pith.science/paper/63BVLHCD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09871&json=true","fetch_graph":"https://pith.science/api/pith-number/63BVLHCD6J25ADKYGTBKPALHDM/graph.json","fetch_events":"https://pith.science/api/pith-number/63BVLHCD6J25ADKYGTBKPALHDM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/63BVLHCD6J25ADKYGTBKPALHDM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/63BVLHCD6J25ADKYGTBKPALHDM/action/storage_attestation","attest_author":"https://pith.science/pith/63BVLHCD6J25ADKYGTBKPALHDM/action/author_attestation","sign_citation":"https://pith.science/pith/63BVLHCD6J25ADKYGTBKPALHDM/action/citation_signature","submit_replication":"https://pith.science/pith/63BVLHCD6J25ADKYGTBKPALHDM/action/replication_record"}},"created_at":"2026-07-05T10:43:40.140488+00:00","updated_at":"2026-07-05T10:43:40.140488+00:00"}