{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XJDR5WX3LR6DY65OJBABTC5YZS","short_pith_number":"pith:XJDR5WX3","schema_version":"1.0","canonical_sha256":"ba471edafb5c7c3c7bae4840198bb8ccb996db241554e7e2485ac2bf55a64c48","source":{"kind":"arxiv","id":"2405.20624","version":1},"attestation_state":"computed","paper":{"title":"Leveraging Large Language Models for Entity Matching","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Qianyu Huang, Tongfang Zhao","submitted_at":"2024-05-31T05:22:07Z","abstract_excerpt":"Entity matching (EM) is a critical task in data integration, aiming to identify records across different datasets that refer to the same real-world entities. Traditional methods often rely on manually engineered features and rule-based systems, which struggle with diverse and unstructured data. The emergence of Large Language Models (LLMs) such as GPT-4 offers transformative potential for EM, leveraging their advanced semantic understanding and contextual capabilities. This vision paper explores the application of LLMs to EM, discussing their advantages, challenges, and future research directi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.20624","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-31T05:22:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f225ba99b58014201725d1c10b5e9590d038741d0aae64a219aa527edd350ce9","abstract_canon_sha256":"4cdb821f96ae65d44968795d198b2e8b22e19c3d7c2cc0c255ba5208413f461e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:39.920309Z","signature_b64":"SHNsxyL01NACBoBQm5mjVVussBZRIzma72lY2PqdBdKrl6jFwICai+HTzw4SbtkyI0nZx4jsIDK2fgMWgaHkBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba471edafb5c7c3c7bae4840198bb8ccb996db241554e7e2485ac2bf55a64c48","last_reissued_at":"2026-07-05T08:25:39.919867Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:39.919867Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Leveraging Large Language Models for Entity Matching","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Qianyu Huang, Tongfang Zhao","submitted_at":"2024-05-31T05:22:07Z","abstract_excerpt":"Entity matching (EM) is a critical task in data integration, aiming to identify records across different datasets that refer to the same real-world entities. Traditional methods often rely on manually engineered features and rule-based systems, which struggle with diverse and unstructured data. The emergence of Large Language Models (LLMs) such as GPT-4 offers transformative potential for EM, leveraging their advanced semantic understanding and contextual capabilities. This vision paper explores the application of LLMs to EM, discussing their advantages, challenges, and future research directi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20624","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.20624/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.20624","created_at":"2026-07-05T08:25:39.919923+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.20624v1","created_at":"2026-07-05T08:25:39.919923+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20624","created_at":"2026-07-05T08:25:39.919923+00:00"},{"alias_kind":"pith_short_12","alias_value":"XJDR5WX3LR6D","created_at":"2026-07-05T08:25:39.919923+00:00"},{"alias_kind":"pith_short_16","alias_value":"XJDR5WX3LR6DY65O","created_at":"2026-07-05T08:25:39.919923+00:00"},{"alias_kind":"pith_short_8","alias_value":"XJDR5WX3","created_at":"2026-07-05T08:25:39.919923+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08096","citing_title":"Identifying unique developers in OSS projects: A family of models","ref_index":135,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XJDR5WX3LR6DY65OJBABTC5YZS","json":"https://pith.science/pith/XJDR5WX3LR6DY65OJBABTC5YZS.json","graph_json":"https://pith.science/api/pith-number/XJDR5WX3LR6DY65OJBABTC5YZS/graph.json","events_json":"https://pith.science/api/pith-number/XJDR5WX3LR6DY65OJBABTC5YZS/events.json","paper":"https://pith.science/paper/XJDR5WX3"},"agent_actions":{"view_html":"https://pith.science/pith/XJDR5WX3LR6DY65OJBABTC5YZS","download_json":"https://pith.science/pith/XJDR5WX3LR6DY65OJBABTC5YZS.json","view_paper":"https://pith.science/paper/XJDR5WX3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.20624&json=true","fetch_graph":"https://pith.science/api/pith-number/XJDR5WX3LR6DY65OJBABTC5YZS/graph.json","fetch_events":"https://pith.science/api/pith-number/XJDR5WX3LR6DY65OJBABTC5YZS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XJDR5WX3LR6DY65OJBABTC5YZS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XJDR5WX3LR6DY65OJBABTC5YZS/action/storage_attestation","attest_author":"https://pith.science/pith/XJDR5WX3LR6DY65OJBABTC5YZS/action/author_attestation","sign_citation":"https://pith.science/pith/XJDR5WX3LR6DY65OJBABTC5YZS/action/citation_signature","submit_replication":"https://pith.science/pith/XJDR5WX3LR6DY65OJBABTC5YZS/action/replication_record"}},"created_at":"2026-07-05T08:25:39.919923+00:00","updated_at":"2026-07-05T08:25:39.919923+00:00"}