{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:YWM3Q7CHKXUGCTTDVXUXOO5ECK","short_pith_number":"pith:YWM3Q7CH","schema_version":"1.0","canonical_sha256":"c599b87c4755e8614e63ade9773ba412b114fb837671d8c6a31af18eb2c29c7f","source":{"kind":"arxiv","id":"2010.12688","version":2},"attestation_state":"computed","paper":{"title":"Knowledge Graph Based Synthetic Corpus Generation for Knowledge-Enhanced Language Model Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Heming Ge, Oshin Agarwal, Rami Al-Rfou, Siamak Shakeri","submitted_at":"2020-10-23T22:14:50Z","abstract_excerpt":"Prior work on Data-To-Text Generation, the task of converting knowledge graph (KG) triples into natural text, focused on domain-specific benchmark datasets. In this paper, however, we verbalize the entire English Wikidata KG, and discuss the unique challenges associated with a broad, open-domain, large-scale verbalization. We further show that verbalizing a comprehensive, encyclopedic KG like Wikidata can be used to integrate structured KGs and natural language corpora. In contrast to the many architectures that have been developed to integrate these two sources, our approach converts the KG i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.12688","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-10-23T22:14:50Z","cross_cats_sorted":[],"title_canon_sha256":"9e1ac86870bbcb5cf277cf25796fbbed95c0bcb98ba2219138923280cdd06857","abstract_canon_sha256":"98af56cf09139db90dd4d2f1feca1d4f5097557633fbe464c7aabaca57cfbe85"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:22:38.961849Z","signature_b64":"gZOmYs4keqsRKjj9iB5l/U4Pn76ZB90D+vjsTh/OD8sXDiKnJSYhSDsOGYZn5cy+uKCdv+Hthk4Ham20SDbEAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c599b87c4755e8614e63ade9773ba412b114fb837671d8c6a31af18eb2c29c7f","last_reissued_at":"2026-07-05T02:22:38.961311Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:22:38.961311Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Knowledge Graph Based Synthetic Corpus Generation for Knowledge-Enhanced Language Model Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Heming Ge, Oshin Agarwal, Rami Al-Rfou, Siamak Shakeri","submitted_at":"2020-10-23T22:14:50Z","abstract_excerpt":"Prior work on Data-To-Text Generation, the task of converting knowledge graph (KG) triples into natural text, focused on domain-specific benchmark datasets. In this paper, however, we verbalize the entire English Wikidata KG, and discuss the unique challenges associated with a broad, open-domain, large-scale verbalization. We further show that verbalizing a comprehensive, encyclopedic KG like Wikidata can be used to integrate structured KGs and natural language corpora. In contrast to the many architectures that have been developed to integrate these two sources, our approach converts the KG i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.12688","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.12688/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.12688","created_at":"2026-07-05T02:22:38.961383+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.12688v2","created_at":"2026-07-05T02:22:38.961383+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.12688","created_at":"2026-07-05T02:22:38.961383+00:00"},{"alias_kind":"pith_short_12","alias_value":"YWM3Q7CHKXUG","created_at":"2026-07-05T02:22:38.961383+00:00"},{"alias_kind":"pith_short_16","alias_value":"YWM3Q7CHKXUGCTTD","created_at":"2026-07-05T02:22:38.961383+00:00"},{"alias_kind":"pith_short_8","alias_value":"YWM3Q7CH","created_at":"2026-07-05T02:22:38.961383+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.00309","citing_title":"Retrieval-Augmented Generation with Graphs (GraphRAG)","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17458","citing_title":"EHRAG: Bridging Semantic Gaps in Lightweight GraphRAG via Hybrid Hypergraph Construction and Retrieval","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YWM3Q7CHKXUGCTTDVXUXOO5ECK","json":"https://pith.science/pith/YWM3Q7CHKXUGCTTDVXUXOO5ECK.json","graph_json":"https://pith.science/api/pith-number/YWM3Q7CHKXUGCTTDVXUXOO5ECK/graph.json","events_json":"https://pith.science/api/pith-number/YWM3Q7CHKXUGCTTDVXUXOO5ECK/events.json","paper":"https://pith.science/paper/YWM3Q7CH"},"agent_actions":{"view_html":"https://pith.science/pith/YWM3Q7CHKXUGCTTDVXUXOO5ECK","download_json":"https://pith.science/pith/YWM3Q7CHKXUGCTTDVXUXOO5ECK.json","view_paper":"https://pith.science/paper/YWM3Q7CH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.12688&json=true","fetch_graph":"https://pith.science/api/pith-number/YWM3Q7CHKXUGCTTDVXUXOO5ECK/graph.json","fetch_events":"https://pith.science/api/pith-number/YWM3Q7CHKXUGCTTDVXUXOO5ECK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YWM3Q7CHKXUGCTTDVXUXOO5ECK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YWM3Q7CHKXUGCTTDVXUXOO5ECK/action/storage_attestation","attest_author":"https://pith.science/pith/YWM3Q7CHKXUGCTTDVXUXOO5ECK/action/author_attestation","sign_citation":"https://pith.science/pith/YWM3Q7CHKXUGCTTDVXUXOO5ECK/action/citation_signature","submit_replication":"https://pith.science/pith/YWM3Q7CHKXUGCTTDVXUXOO5ECK/action/replication_record"}},"created_at":"2026-07-05T02:22:38.961383+00:00","updated_at":"2026-07-05T02:22:38.961383+00:00"}