{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MFDTK35I43FK4U67F3HOZPJGAV","short_pith_number":"pith:MFDTK35I","schema_version":"1.0","canonical_sha256":"6147356fa8e6caae53df2eceecbd26057d3be17b77903e32fbf235edfdcfa85e","source":{"kind":"arxiv","id":"2402.14710","version":3},"attestation_state":"computed","paper":{"title":"IEPile: Unearthing Large-Scale Schema-Based Information Extraction Corpus","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.DB","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hongbin Ye, Honghao Gui, Huajun Chen, Lei Liang, Lin Yuan, Mengshu Sun, Ningyu Zhang","submitted_at":"2024-02-22T17:11:38Z","abstract_excerpt":"Large Language Models (LLMs) demonstrate remarkable potential across various domains; however, they exhibit a significant performance gap in Information Extraction (IE). Note that high-quality instruction data is the vital key for enhancing the specific capabilities of LLMs, while current IE datasets tend to be small in scale, fragmented, and lack standardized schema. To this end, we introduce IEPile, a comprehensive bilingual (English and Chinese) IE instruction corpus, which contains approximately 0.32B tokens. We construct IEPile by collecting and cleaning 33 existing IE datasets, and intro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14710","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-22T17:11:38Z","cross_cats_sorted":["cs.AI","cs.DB","cs.IR","cs.LG"],"title_canon_sha256":"2e7dedcd2aadc62360d4b7ba716e9013dc2e5f656e865da16847d701ecae9e09","abstract_canon_sha256":"8b0069b27a1485e164ad90791dfd3122af04ff2f387299cc19da3dc75e1a6ae6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:18.459863Z","signature_b64":"sw30RNvOozWGUeFyeEwFyx4xpyJYbAhscQDamLhD096iNCL/KHTnf4VyaxDjB+tCkkehzhqT5oVMwOGF49HqAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6147356fa8e6caae53df2eceecbd26057d3be17b77903e32fbf235edfdcfa85e","last_reissued_at":"2026-07-05T08:23:18.459410Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:18.459410Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"IEPile: Unearthing Large-Scale Schema-Based Information Extraction Corpus","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.DB","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hongbin Ye, Honghao Gui, Huajun Chen, Lei Liang, Lin Yuan, Mengshu Sun, Ningyu Zhang","submitted_at":"2024-02-22T17:11:38Z","abstract_excerpt":"Large Language Models (LLMs) demonstrate remarkable potential across various domains; however, they exhibit a significant performance gap in Information Extraction (IE). Note that high-quality instruction data is the vital key for enhancing the specific capabilities of LLMs, while current IE datasets tend to be small in scale, fragmented, and lack standardized schema. To this end, we introduce IEPile, a comprehensive bilingual (English and Chinese) IE instruction corpus, which contains approximately 0.32B tokens. We construct IEPile by collecting and cleaning 33 existing IE datasets, and intro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14710","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14710/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14710","created_at":"2026-07-05T08:23:18.459466+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14710v3","created_at":"2026-07-05T08:23:18.459466+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14710","created_at":"2026-07-05T08:23:18.459466+00:00"},{"alias_kind":"pith_short_12","alias_value":"MFDTK35I43FK","created_at":"2026-07-05T08:23:18.459466+00:00"},{"alias_kind":"pith_short_16","alias_value":"MFDTK35I43FK4U67","created_at":"2026-07-05T08:23:18.459466+00:00"},{"alias_kind":"pith_short_8","alias_value":"MFDTK35I","created_at":"2026-07-05T08:23:18.459466+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.06881","citing_title":"KnowCoder-V2: Deep Knowledge Analysis","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MFDTK35I43FK4U67F3HOZPJGAV","json":"https://pith.science/pith/MFDTK35I43FK4U67F3HOZPJGAV.json","graph_json":"https://pith.science/api/pith-number/MFDTK35I43FK4U67F3HOZPJGAV/graph.json","events_json":"https://pith.science/api/pith-number/MFDTK35I43FK4U67F3HOZPJGAV/events.json","paper":"https://pith.science/paper/MFDTK35I"},"agent_actions":{"view_html":"https://pith.science/pith/MFDTK35I43FK4U67F3HOZPJGAV","download_json":"https://pith.science/pith/MFDTK35I43FK4U67F3HOZPJGAV.json","view_paper":"https://pith.science/paper/MFDTK35I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14710&json=true","fetch_graph":"https://pith.science/api/pith-number/MFDTK35I43FK4U67F3HOZPJGAV/graph.json","fetch_events":"https://pith.science/api/pith-number/MFDTK35I43FK4U67F3HOZPJGAV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MFDTK35I43FK4U67F3HOZPJGAV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MFDTK35I43FK4U67F3HOZPJGAV/action/storage_attestation","attest_author":"https://pith.science/pith/MFDTK35I43FK4U67F3HOZPJGAV/action/author_attestation","sign_citation":"https://pith.science/pith/MFDTK35I43FK4U67F3HOZPJGAV/action/citation_signature","submit_replication":"https://pith.science/pith/MFDTK35I43FK4U67F3HOZPJGAV/action/replication_record"}},"created_at":"2026-07-05T08:23:18.459466+00:00","updated_at":"2026-07-05T08:23:18.459466+00:00"}