{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6QGL7VVTQSJRBZURC6DWEFL75N","short_pith_number":"pith:6QGL7VVT","schema_version":"1.0","canonical_sha256":"f40cbfd6b3849310e691178762157feb73944d814a4ed4d7c16b626d6db79d6f","source":{"kind":"arxiv","id":"2406.02962","version":1},"attestation_state":"computed","paper":{"title":"Docs2KG: Unified Knowledge Graph Construction from Heterogeneous Documents Assisted by Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Jichunyang Li, Kai Niu, Qiang Sun, Sirui Li, Wei Liu, Wenxiao Zhang, Xiangrui Kong, Yuanyi Luo","submitted_at":"2024-06-05T05:35:59Z","abstract_excerpt":"Even for a conservative estimate, 80% of enterprise data reside in unstructured files, stored in data lakes that accommodate heterogeneous formats. Classical search engines can no longer meet information seeking needs, especially when the task is to browse and explore for insight formulation. In other words, there are no obvious search keywords to use. Knowledge graphs, due to their natural visual appeals that reduce the human cognitive load, become the winning candidate for heterogeneous data integration and knowledge representation.\n  In this paper, we introduce Docs2KG, a novel framework de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.02962","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-05T05:35:59Z","cross_cats_sorted":["cs.AI","cs.IR"],"title_canon_sha256":"14ca92a790b6664ae32d624afaee873b1b408e69fade928a7e6f12cfa8e654ad","abstract_canon_sha256":"52632646384795c7cf9535f81c3b0c7c0d23b1b88d182e82dbca718dec4f0d97"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:33.372372Z","signature_b64":"vVNsqHjhZ4yeD0/JW5aeR3vmDYNyFMVUAiNZaZzVi5Bo26j5aYaaakD12GlQKE6Wln+suFEBqIRDmuVRhNkcAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f40cbfd6b3849310e691178762157feb73944d814a4ed4d7c16b626d6db79d6f","last_reissued_at":"2026-07-05T08:27:33.371831Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:33.371831Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Docs2KG: Unified Knowledge Graph Construction from Heterogeneous Documents Assisted by Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Jichunyang Li, Kai Niu, Qiang Sun, Sirui Li, Wei Liu, Wenxiao Zhang, Xiangrui Kong, Yuanyi Luo","submitted_at":"2024-06-05T05:35:59Z","abstract_excerpt":"Even for a conservative estimate, 80% of enterprise data reside in unstructured files, stored in data lakes that accommodate heterogeneous formats. Classical search engines can no longer meet information seeking needs, especially when the task is to browse and explore for insight formulation. In other words, there are no obvious search keywords to use. Knowledge graphs, due to their natural visual appeals that reduce the human cognitive load, become the winning candidate for heterogeneous data integration and knowledge representation.\n  In this paper, we introduce Docs2KG, a novel framework de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.02962","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.02962/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.02962","created_at":"2026-07-05T08:27:33.371888+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.02962v1","created_at":"2026-07-05T08:27:33.371888+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.02962","created_at":"2026-07-05T08:27:33.371888+00:00"},{"alias_kind":"pith_short_12","alias_value":"6QGL7VVTQSJR","created_at":"2026-07-05T08:27:33.371888+00:00"},{"alias_kind":"pith_short_16","alias_value":"6QGL7VVTQSJRBZUR","created_at":"2026-07-05T08:27:33.371888+00:00"},{"alias_kind":"pith_short_8","alias_value":"6QGL7VVT","created_at":"2026-07-05T08:27:33.371888+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05724","citing_title":"Narrative Knowledge Weaver: Narrative-Centric Retrieval-Augmented Reasoning for Long-Form Text Understanding","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04948","citing_title":"From PDF to RAG-Ready: Evaluating Document Conversion Frameworks for Domain-Specific Question Answering","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02477","citing_title":"Guideline2Graph: Profile-Aware Multimodal Parsing for Executable Clinical Decision Graphs","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6QGL7VVTQSJRBZURC6DWEFL75N","json":"https://pith.science/pith/6QGL7VVTQSJRBZURC6DWEFL75N.json","graph_json":"https://pith.science/api/pith-number/6QGL7VVTQSJRBZURC6DWEFL75N/graph.json","events_json":"https://pith.science/api/pith-number/6QGL7VVTQSJRBZURC6DWEFL75N/events.json","paper":"https://pith.science/paper/6QGL7VVT"},"agent_actions":{"view_html":"https://pith.science/pith/6QGL7VVTQSJRBZURC6DWEFL75N","download_json":"https://pith.science/pith/6QGL7VVTQSJRBZURC6DWEFL75N.json","view_paper":"https://pith.science/paper/6QGL7VVT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.02962&json=true","fetch_graph":"https://pith.science/api/pith-number/6QGL7VVTQSJRBZURC6DWEFL75N/graph.json","fetch_events":"https://pith.science/api/pith-number/6QGL7VVTQSJRBZURC6DWEFL75N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6QGL7VVTQSJRBZURC6DWEFL75N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6QGL7VVTQSJRBZURC6DWEFL75N/action/storage_attestation","attest_author":"https://pith.science/pith/6QGL7VVTQSJRBZURC6DWEFL75N/action/author_attestation","sign_citation":"https://pith.science/pith/6QGL7VVTQSJRBZURC6DWEFL75N/action/citation_signature","submit_replication":"https://pith.science/pith/6QGL7VVTQSJRBZURC6DWEFL75N/action/replication_record"}},"created_at":"2026-07-05T08:27:33.371888+00:00","updated_at":"2026-07-05T08:27:33.371888+00:00"}