{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CM2B5KC3DMLRN6W53QZWF7UN7Z","short_pith_number":"pith:CM2B5KC3","schema_version":"1.0","canonical_sha256":"13341ea85b1b1716fadddc3362fe8dfe59aea38a1c991d509965db8a3497505a","source":{"kind":"arxiv","id":"2503.21810","version":1},"attestation_state":"computed","paper":{"title":"Taxonomy Inference for Tabular Data Using Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.IR"],"primary_cat":"cs.DB","authors_text":"Jiaoyan Chen, Norman W. Paton, Zhenyu Wu","submitted_at":"2025-03-25T16:26:05Z","abstract_excerpt":"Taxonomy inference for tabular data is a critical task of schema inference, aiming at discovering entity types (i.e., concepts) of the tables and building their hierarchy. It can play an important role in data management, data exploration, ontology learning, and many data-centric applications. Existing schema inference systems focus more on XML, JSON or RDF data, and often rely on lexical formats and structures of the data for calculating similarities, with limited exploitation of the semantics of the text across a table. Motivated by recent works on taxonomy completion and construction using "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.21810","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DB","submitted_at":"2025-03-25T16:26:05Z","cross_cats_sorted":["cs.AI","cs.CL","cs.IR"],"title_canon_sha256":"cbc5655f9eadec5c7dcd7f7f48e5199ba58f7de139d55372e289eb5ca5e22154","abstract_canon_sha256":"782081f336eb8ad074c5640af77d423ab2d0fb94b99e0bf65d748ca4357f22be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:02.593963Z","signature_b64":"cawQL8V34DZvLoF0Vg9pxBpW/1uYfBJf8EZcd36P8cNzaAl8XuFpo5ATyF7dn8x/Hq7yB4b4WpWLJnv784GhCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"13341ea85b1b1716fadddc3362fe8dfe59aea38a1c991d509965db8a3497505a","last_reissued_at":"2026-07-05T10:41:02.593484Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:02.593484Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Taxonomy Inference for Tabular Data Using Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.IR"],"primary_cat":"cs.DB","authors_text":"Jiaoyan Chen, Norman W. Paton, Zhenyu Wu","submitted_at":"2025-03-25T16:26:05Z","abstract_excerpt":"Taxonomy inference for tabular data is a critical task of schema inference, aiming at discovering entity types (i.e., concepts) of the tables and building their hierarchy. It can play an important role in data management, data exploration, ontology learning, and many data-centric applications. Existing schema inference systems focus more on XML, JSON or RDF data, and often rely on lexical formats and structures of the data for calculating similarities, with limited exploitation of the semantics of the text across a table. Motivated by recent works on taxonomy completion and construction using "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.21810","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.21810/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.21810","created_at":"2026-07-05T10:41:02.593543+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.21810v1","created_at":"2026-07-05T10:41:02.593543+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.21810","created_at":"2026-07-05T10:41:02.593543+00:00"},{"alias_kind":"pith_short_12","alias_value":"CM2B5KC3DMLR","created_at":"2026-07-05T10:41:02.593543+00:00"},{"alias_kind":"pith_short_16","alias_value":"CM2B5KC3DMLRN6W5","created_at":"2026-07-05T10:41:02.593543+00:00"},{"alias_kind":"pith_short_8","alias_value":"CM2B5KC3","created_at":"2026-07-05T10:41:02.593543+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.21056","citing_title":"AI-Driven Generation of Data Contracts in Modern Data Engineering Systems","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CM2B5KC3DMLRN6W53QZWF7UN7Z","json":"https://pith.science/pith/CM2B5KC3DMLRN6W53QZWF7UN7Z.json","graph_json":"https://pith.science/api/pith-number/CM2B5KC3DMLRN6W53QZWF7UN7Z/graph.json","events_json":"https://pith.science/api/pith-number/CM2B5KC3DMLRN6W53QZWF7UN7Z/events.json","paper":"https://pith.science/paper/CM2B5KC3"},"agent_actions":{"view_html":"https://pith.science/pith/CM2B5KC3DMLRN6W53QZWF7UN7Z","download_json":"https://pith.science/pith/CM2B5KC3DMLRN6W53QZWF7UN7Z.json","view_paper":"https://pith.science/paper/CM2B5KC3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.21810&json=true","fetch_graph":"https://pith.science/api/pith-number/CM2B5KC3DMLRN6W53QZWF7UN7Z/graph.json","fetch_events":"https://pith.science/api/pith-number/CM2B5KC3DMLRN6W53QZWF7UN7Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CM2B5KC3DMLRN6W53QZWF7UN7Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CM2B5KC3DMLRN6W53QZWF7UN7Z/action/storage_attestation","attest_author":"https://pith.science/pith/CM2B5KC3DMLRN6W53QZWF7UN7Z/action/author_attestation","sign_citation":"https://pith.science/pith/CM2B5KC3DMLRN6W53QZWF7UN7Z/action/citation_signature","submit_replication":"https://pith.science/pith/CM2B5KC3DMLRN6W53QZWF7UN7Z/action/replication_record"}},"created_at":"2026-07-05T10:41:02.593543+00:00","updated_at":"2026-07-05T10:41:02.593543+00:00"}